@tea-agent/loop-agent 0.39.0-beta.12 → 0.39.0-beta.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +27 -0
- package/dist/application/dag/generate-task-dag.js +6 -2
- package/dist/application/task-lifecycle/advance.js +20 -4
- package/dist/application/task-lifecycle/observe.js +171 -17
- package/dist/application/task-lifecycle/plan-transitions.js +42 -7
- package/dist/build-stamp.json +3 -3
- package/dist/commands/client-recovery.js +8 -36
- package/dist/executors/dag-pi-executor.js +636 -109
- package/dist/executors/pi-executor.js +8 -4
- package/dist/executors/pi-sdk-executor.js +33 -5
- package/dist/executors/shell-executor.js +102 -32
- package/dist/governance/checks.js +1 -0
- package/dist/shared/pi-context-pressure/checkpoint.js +116 -0
- package/dist/shared/pi-context-pressure/compaction-policy.js +151 -0
- package/dist/shared/pi-context-pressure/env.js +58 -0
- package/dist/shared/pi-context-pressure/extension.js +100 -0
- package/dist/shared/pi-context-pressure/index.js +7 -0
- package/dist/shared/pi-context-pressure/overflow.js +252 -0
- package/dist/shared/pi-context-pressure/sift-bridge.js +386 -0
- package/dist/shared/pi-context-pressure/telemetry.js +51 -0
- package/dist/task/frontend-project-capability.js +3 -1
- package/dist/task/source-prepare/fragment-inventory.js +4 -1
- package/dist/worker/console/chat/pi-runtime.js +146 -5
- package/dist/worker/console/chat/provider-error.js +2 -1
- package/dist/worker/console/chat/routes.js +3 -0
- package/dist/worker/console/chat/sift-bridge.js +1 -0
- package/dist/worker/console/dag-execution-receipt.js +20 -2
- package/dist/worker/console/operator-actions.js +4 -3
- package/dist/worker/console/static/assets/{abnfDiagram-N423BO3Z-CXj_GnSb.js → abnfDiagram-N423BO3Z-DC863mud.js} +1 -1
- package/dist/worker/console/static/assets/{arc-BZp6JAp7.js → arc-CftC38G9.js} +1 -1
- package/dist/worker/console/static/assets/{architectureDiagram-T3A2C74G-DfpcEuYU.js → architectureDiagram-T3A2C74G-B3PAPQCw.js} +1 -1
- package/dist/worker/console/static/assets/{blockDiagram-VBNYF7ZC-_Gf0xadb.js → blockDiagram-VBNYF7ZC-C9UDJNRv.js} +1 -1
- package/dist/worker/console/static/assets/{c4Diagram-5PPSVZJV-Ct2QPCmv.js → c4Diagram-5PPSVZJV--BJ76fv_.js} +1 -1
- package/dist/worker/console/static/assets/channel-CUz-Bg86.js +1 -0
- package/dist/worker/console/static/assets/{chunk-2GRJ4B5K-BfklkzKl.js → chunk-2GRJ4B5K--hiIqoGp.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-2Q5K7J3B-Dy25vZJV.js → chunk-2Q5K7J3B-DgTzpIa3.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-5RXB4S5H-BFlCRZep.js → chunk-5RXB4S5H-DIwpJziP.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-5VM5RSS4-op3oVxIE.js → chunk-5VM5RSS4-BJzbdUi0.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-6Q2QTUOP-C0R7rzF2.js → chunk-6Q2QTUOP-BbGouI1Z.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-GF5L2VYU-Bt44TCGy.js → chunk-GF5L2VYU-B9247w69.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-JWPE2WC7-_uEE_XFx.js → chunk-JWPE2WC7-CXob_wYy.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-KBJHAD2P-C3TOYGZ9.js → chunk-KBJHAD2P-C6FUel1g.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-RYQCIY6F-Cz60oBPV.js → chunk-RYQCIY6F-DGKRYKHu.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-XXDRQBXY-8Bik0qis.js → chunk-XXDRQBXY-kNBqNPfx.js} +1 -1
- package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-BhlUacmD.js +1 -0
- package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-BhlUacmD.js +1 -0
- package/dist/worker/console/static/assets/{cose-bilkent-JH36ORCC-C2QIOA4a.js → cose-bilkent-JH36ORCC--xmkDwfD.js} +1 -1
- package/dist/worker/console/static/assets/{cynefin-VYW2F7L2-CSH_yUUd.js → cynefin-VYW2F7L2-Cw5FMMuG.js} +1 -1
- package/dist/worker/console/static/assets/{cynefinDiagram-MW4NZA55-BDLw2XFM.js → cynefinDiagram-MW4NZA55-CuLslt2t.js} +1 -1
- package/dist/worker/console/static/assets/{dagre-VZM6K2ZE-CcD9ZtF4.js → dagre-VZM6K2ZE-D9ngqrs2.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-7IWD3JNH-DqwtkBsI.js → diagram-7IWD3JNH-RDRSbHQp.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-B4RE2ZJO-BWVEcqsC.js → diagram-B4RE2ZJO-BgQ1P9aV.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-LBJQPF4R-M-WAbtQi.js → diagram-LBJQPF4R-D3WWyPtn.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-Q27KOJAE-DKW7SixP.js → diagram-Q27KOJAE-L2k28wTR.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-UB23O5K3-PHHPPtrj.js → diagram-UB23O5K3-DDNrA5Qw.js} +1 -1
- package/dist/worker/console/static/assets/{ebnfDiagram-BXEA7PRR-BQD-B4RX.js → ebnfDiagram-BXEA7PRR-BaRFCOdM.js} +1 -1
- package/dist/worker/console/static/assets/{erDiagram-JOGREHBK-BFvULb51.js → erDiagram-JOGREHBK-CM33DVVM.js} +1 -1
- package/dist/worker/console/static/assets/{flowDiagram-UKHOOZJN-Bw-aXTCb.js → flowDiagram-UKHOOZJN-CVBSjOxe.js} +1 -1
- package/dist/worker/console/static/assets/{ganttDiagram-PKOTCBZU-BGKw04Qx.js → ganttDiagram-PKOTCBZU-BdscewE_.js} +1 -1
- package/dist/worker/console/static/assets/{gitGraphDiagram-DS77QQ5N-RqKaPR-I.js → gitGraphDiagram-DS77QQ5N-HRuSZb7g.js} +1 -1
- package/dist/worker/console/static/assets/{index-B_D8rbWc.js → index-B9JJQsVK.js} +102 -72
- package/dist/worker/console/static/assets/index-rWaGv4jz.css +1 -0
- package/dist/worker/console/static/assets/{infoDiagram-6WML65LV-YO5dnzrJ.js → infoDiagram-6WML65LV-cYfHvfAR.js} +1 -1
- package/dist/worker/console/static/assets/{ishikawaDiagram-WSZJBQD7-Bi1VZHMq.js → ishikawaDiagram-WSZJBQD7-CZmaoOuy.js} +1 -1
- package/dist/worker/console/static/assets/{journeyDiagram-NVQOT4AX-Bszls1DD.js → journeyDiagram-NVQOT4AX-ntYijh7g.js} +1 -1
- package/dist/worker/console/static/assets/{kanban-definition-27J2QSJJ-PhzeaZ09.js → kanban-definition-27J2QSJJ-Cz3b0DeD.js} +1 -1
- package/dist/worker/console/static/assets/{linear-BsjbDoXi.js → linear-DvGonpsP.js} +1 -1
- package/dist/worker/console/static/assets/{mermaid.core-0B7NnWKk.js → mermaid.core-B-3vjfyW.js} +5 -5
- package/dist/worker/console/static/assets/{mindmap-definition-FAOFIHXS-BJr4Fj-q.js → mindmap-definition-FAOFIHXS-5iKxlsX3.js} +1 -1
- package/dist/worker/console/static/assets/{pegDiagram-VL7TDLO6-moC4fpGB.js → pegDiagram-VL7TDLO6-C8BUwapW.js} +1 -1
- package/dist/worker/console/static/assets/{pieDiagram-7S7Q4E2Y-BEw37-2c.js → pieDiagram-7S7Q4E2Y-B2rPoQQG.js} +1 -1
- package/dist/worker/console/static/assets/{quadrantDiagram-CIZ2JOQS-Cq6LyasU.js → quadrantDiagram-CIZ2JOQS-jDeVUAy4.js} +1 -1
- package/dist/worker/console/static/assets/{railroadDiagram-AXF67PYL-DUCMcK0D.js → railroadDiagram-AXF67PYL-DV140Dwm.js} +1 -1
- package/dist/worker/console/static/assets/{requirementDiagram-LRYGKXZP-C3upTZm7.js → requirementDiagram-LRYGKXZP-CnYgHtYC.js} +1 -1
- package/dist/worker/console/static/assets/{sankeyDiagram-W5VNT64P-BI_gMsCW.js → sankeyDiagram-W5VNT64P-SIK3CdWw.js} +1 -1
- package/dist/worker/console/static/assets/{sequenceDiagram-SI44F4Z6-YFOIRzfN.js → sequenceDiagram-SI44F4Z6-DW_vZix7.js} +1 -1
- package/dist/worker/console/static/assets/{sizeCapture-X5ZJPWSS-dOnB7UDD.js → sizeCapture-X5ZJPWSS-NBAtkg0C.js} +1 -1
- package/dist/worker/console/static/assets/{stateDiagram-OKZ733FA-BXUniaIh.js → stateDiagram-OKZ733FA-Bi1bQxpi.js} +1 -1
- package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-DjWKYjQ1.js +1 -0
- package/dist/worker/console/static/assets/{swimlanes-SLNWSIFB-2tA4wTNu.js → swimlanes-SLNWSIFB-8PT_uP_i.js} +2 -2
- package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-B4cdm7bH.js +8 -0
- package/dist/worker/console/static/assets/{timeline-definition-Z64GVDOM-DO0HkJXC.js → timeline-definition-Z64GVDOM-CD0ZZPk0.js} +1 -1
- package/dist/worker/console/static/assets/{vennDiagram-T6HMQDX7-DvIOzixv.js → vennDiagram-T6HMQDX7-BbEgWdhK.js} +1 -1
- package/dist/worker/console/static/assets/{wardleyDiagram-T6FBY63Y-DXDZ0cTj.js → wardleyDiagram-T6FBY63Y-DCwPKdLl.js} +1 -1
- package/dist/worker/console/static/assets/{xychartDiagram-ELKLHX3M-B32Ark0D.js → xychartDiagram-ELKLHX3M-sfC5PcNg.js} +1 -1
- package/dist/worker/console/static/index.html +2 -2
- package/dist/worker/observe/static/styles.css +9 -0
- package/dist/worker/observe/static/views/dag-inspector.js +40 -0
- package/dist/worker/observe/static/views/session-timeline.js +135 -0
- package/dist/workflows/dag/backend-test-case-coverage-analysis.js +157 -7
- package/dist/workflows/dag/backend-test-pytest-collection.js +70 -2
- package/dist/workflows/dag/backend-test-result-contract.js +4 -0
- package/dist/workflows/dag/backend-test-scenario-param.js +339 -53
- package/dist/workflows/dag/backend-test-writer-completeness.js +11 -0
- package/dist/workflows/dag/dag-retry-schema.js +138 -0
- package/dist/workflows/dag/frontend-implementation-contract.js +174 -31
- package/dist/workflows/dag/frontend-review-context.js +12 -1
- package/dist/workflows/dag/frontend-shadow-dual-write.js +59 -13
- package/dist/workflows/dag/frontend-writer-admission.js +13 -0
- package/dist/workflows/dag/init-hybrid.js +8 -3
- package/dist/workflows/dag/node-execution.js +277 -3
- package/dist/workflows/dag/rerun-feedback.js +135 -3
- package/dist/workflows/dag/retry-policy.js +13 -122
- package/dist/workflows/dag/types.js +11 -4
- package/docs/architecture/runtime-boundaries.md +2 -1
- package/docs/templates/backend-test-dag.json +4 -3
- package/harness.json +3 -3
- package/package.json +4 -3
- package/dist/worker/console/static/assets/channel-3TxJgYaH.js +0 -1
- package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-BHIkXpp3.js +0 -1
- package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-BHIkXpp3.js +0 -1
- package/dist/worker/console/static/assets/index-BdNx6fj0.css +0 -1
- package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-Cdi6UhLa.js +0 -1
- package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-D-RJBbb0.js +0 -8
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import path from "node:path";
|
|
2
2
|
import { createHash, randomUUID } from "node:crypto";
|
|
3
|
-
import { readFile } from "node:fs/promises";
|
|
3
|
+
import { readFile, stat } from "node:fs/promises";
|
|
4
4
|
import { writeDagNodeJsonArtifact, writeTextArtifactFile, } from "../infrastructure/harness/artifact-store.js";
|
|
5
5
|
import { writeJsonAtomic } from "../infrastructure/harness/atomic-write.js";
|
|
6
6
|
import { mapContractBlockedOwner } from "../workflows/dag/frontend-human-decision.js";
|
|
@@ -15,6 +15,8 @@ import { parseLedgerJson } from "../task/source-prepare/ledger.js";
|
|
|
15
15
|
import { redactPromptForLog, truncateOutput, } from "../shared/output-truncation.js";
|
|
16
16
|
import { GitStatusUnavailableError, pathsChangedDuringRun, readGitStatusPorcelain, recoverRootNulArtifact, snapshotGitStatusPathFingerprints, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
|
|
17
17
|
import { captureWorkspaceWriteSnapshot, diffWorkspaceWriteSnapshots, } from "./workspace-write-snapshot.js";
|
|
18
|
+
import { pathMatchesPattern } from "../shared/git-progress.js";
|
|
19
|
+
import { isSuspiciousVerificationSymbol } from "../workflows/dag/frontend-implementation-contract.js";
|
|
18
20
|
import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
|
|
19
21
|
import { assessBackendTestMdPlanCompleteness, assessBackendTestMdWriterCompleteness, assessBackendTestPytestPlanCompleteness, assessBackendTestPytestWriterCompleteness, assessBackendTestShardChildCompleteness, backendTestWriterProgressRoleForTask, classifyBackendTestWriterCompletenessFailure, isBackendTestCompletenessRetryCandidate, isBackendTestMdPlanTask, isBackendTestPytestPlanTask, isBackendTestShardChildTask, writeBackendTestWriterProgressArtifacts, } from "../workflows/dag/backend-test-writer-completeness.js";
|
|
20
22
|
import { resolveBackendTestLayout } from "../workflows/dag/backend-test-layout.js";
|
|
@@ -261,6 +263,45 @@ export function isFrontendScoutEvidenceNode(task) {
|
|
|
261
263
|
export function isFrontendPlanLedgerNode(task) {
|
|
262
264
|
return task.id === "frontend-plan-pi";
|
|
263
265
|
}
|
|
266
|
+
/** Facts-first nodes whose authoritative output is a committed typed terminal
|
|
267
|
+
* fact (not the assistant text). Downstream compilation reads the flushed
|
|
268
|
+
* `<nodeId>/<file>` store and never the node narrative, so a committed
|
|
269
|
+
* terminal means the work is done. */
|
|
270
|
+
const TYPED_TERMINAL_FACT_NODES = {
|
|
271
|
+
"frontend-contract-pi": {
|
|
272
|
+
file: "contract-typed-facts.jsonl",
|
|
273
|
+
kind: "contract-finalized",
|
|
274
|
+
},
|
|
275
|
+
"frontend-plan-pi": {
|
|
276
|
+
file: "plan-typed-facts.jsonl",
|
|
277
|
+
kind: "finalize_plan",
|
|
278
|
+
},
|
|
279
|
+
};
|
|
280
|
+
/**
|
|
281
|
+
* Accept a facts-terminal node result whose final assistant text is blank
|
|
282
|
+
* when the typed terminal fact was committed successfully. Small-output
|
|
283
|
+
* models legitimately end after the terminal tool call; without this the
|
|
284
|
+
* empty assistantText fails the node as empty-output, the failure classifier
|
|
285
|
+
* phrase-scans the whole session stream and can mislabel the committed run
|
|
286
|
+
* as rate-limit/network, and the finished ledger is thrown away for a
|
|
287
|
+
* deterministic retry that burns the full prompt budget again. Fail-closed:
|
|
288
|
+
* acceptance requires a committed terminal record from the flushed typed
|
|
289
|
+
* facts store; provider-error attempts (non-empty stderr) are never accepted
|
|
290
|
+
* by the caller.
|
|
291
|
+
*/
|
|
292
|
+
export async function acceptCommittedTypedTerminalFact(runDir, nodeId) {
|
|
293
|
+
const binding = TYPED_TERMINAL_FACT_NODES[nodeId];
|
|
294
|
+
if (!binding)
|
|
295
|
+
return false;
|
|
296
|
+
try {
|
|
297
|
+
const { readCommittedOriginFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
|
|
298
|
+
const records = await readCommittedOriginFacts(runDir, nodeId, binding.file);
|
|
299
|
+
return records.some((record) => record.fact.kind === binding.kind);
|
|
300
|
+
}
|
|
301
|
+
catch {
|
|
302
|
+
return false;
|
|
303
|
+
}
|
|
304
|
+
}
|
|
264
305
|
export function resolveDagPiToolNames(task) {
|
|
265
306
|
if (isFrontendReviewTypedTerminalNode(task)) {
|
|
266
307
|
return [
|
|
@@ -277,8 +318,11 @@ export function resolveDagPiToolNames(task) {
|
|
|
277
318
|
];
|
|
278
319
|
}
|
|
279
320
|
if (isFrontendContractTypedNode(task)) {
|
|
321
|
+
// Contract is an incremental-commit node: the source-fidelity ledger is
|
|
322
|
+
// compiled into the <frontend_contract_input> block (node-execution), so
|
|
323
|
+
// no read tools — mirrors the plan node. Omitting read tools prevents a
|
|
324
|
+
// contract from spending its output budget re-reading the raw source.
|
|
280
325
|
return [
|
|
281
|
-
...DAG_PI_READONLY_TOOLS,
|
|
282
326
|
...FRONTEND_CONTRACT_RECORD_TOOL_NAMES,
|
|
283
327
|
...FRONTEND_CONTRACT_TERMINAL_TOOL_NAMES,
|
|
284
328
|
];
|
|
@@ -342,10 +386,6 @@ export function scanReviewTerminalKindsFromSessionEvents(content) {
|
|
|
342
386
|
}
|
|
343
387
|
return kinds;
|
|
344
388
|
}
|
|
345
|
-
/** Read-only discovery tool budget for frontend-plan-pi. Exceeding it means
|
|
346
|
-
* the plan re-read upstream outputs/sources instead of trusting typed facts,
|
|
347
|
-
* which blows up the context window (400 request-too-large). */
|
|
348
|
-
const PLAN_READ_TOOL_BUDGET = 40;
|
|
349
389
|
const READ_ONLY_TOOL_NAMES = new Set(["read", "grep", "ls", "find"]);
|
|
350
390
|
function readEventPath(event) {
|
|
351
391
|
const candidates = [event.path, event.readPath, event.input];
|
|
@@ -430,45 +470,6 @@ export async function detectNodeReadBudget(input) {
|
|
|
430
470
|
issues.push(`frontend ${input.nodeId} read budget exceeded: ${stats.elapsedMs}ms (budget ${input.budget.maxMs}ms)`);
|
|
431
471
|
return issues;
|
|
432
472
|
}
|
|
433
|
-
/** Deterministic read-burst guard for frontend-plan-pi: count read-only
|
|
434
|
-
* discovery tool calls (read/grep/ls/find) from the session log. Over budget →
|
|
435
|
-
* read-burst, retried with a reduced-reading instruction. Pure scan; a
|
|
436
|
-
* successful plan under budget is never blocked. */
|
|
437
|
-
export async function detectPlanReadBurst(input) {
|
|
438
|
-
const sessionEventsPath = path.join(input.runDir, input.nodeId, "session-events.jsonl");
|
|
439
|
-
let count = 0;
|
|
440
|
-
try {
|
|
441
|
-
const content = await readFile(sessionEventsPath, "utf8");
|
|
442
|
-
for (const line of content.split("\n")) {
|
|
443
|
-
if (!line.trim())
|
|
444
|
-
continue;
|
|
445
|
-
try {
|
|
446
|
-
const event = JSON.parse(line);
|
|
447
|
-
if (event.type === "tool_execution_start" &&
|
|
448
|
-
typeof event.toolName === "string" &&
|
|
449
|
-
(event.toolName === "read" ||
|
|
450
|
-
event.toolName === "grep" ||
|
|
451
|
-
event.toolName === "ls" ||
|
|
452
|
-
event.toolName === "find")) {
|
|
453
|
-
count += 1;
|
|
454
|
-
}
|
|
455
|
-
}
|
|
456
|
-
catch {
|
|
457
|
-
// skip unparseable line
|
|
458
|
-
}
|
|
459
|
-
}
|
|
460
|
-
}
|
|
461
|
-
catch {
|
|
462
|
-
// Missing/unreadable session log → no burst detection
|
|
463
|
-
return [];
|
|
464
|
-
}
|
|
465
|
-
if (count > PLAN_READ_TOOL_BUDGET) {
|
|
466
|
-
return [
|
|
467
|
-
`frontend plan read-burst: ${count} read-only tool calls (budget ${PLAN_READ_TOOL_BUDGET}). Trust the upstream contract/scout typed facts; do not re-read contract/scout outputs or source files already captured. Minimize discovery reads, commit record_* facts directly, then finalize_plan.`,
|
|
468
|
-
];
|
|
469
|
-
}
|
|
470
|
-
return [];
|
|
471
|
-
}
|
|
472
473
|
export const FRONTEND_DESIGN_TERMINAL_TOOL_NAMES = new Set([
|
|
473
474
|
"approve_design",
|
|
474
475
|
"request_design_changes",
|
|
@@ -829,8 +830,9 @@ async function loadContractRequirementInheritance(runDir) {
|
|
|
829
830
|
}
|
|
830
831
|
/**
|
|
831
832
|
* Resolve task-source citations from the source-fidelity ledger before the
|
|
832
|
-
* planner starts. The planner names a frozen requirement id
|
|
833
|
-
* to re-read a PRD merely to recover a
|
|
833
|
+
* planner starts. The planner names a frozen requirement id and one of its
|
|
834
|
+
* fragment ids; it never needs to re-read a PRD merely to recover a
|
|
835
|
+
* path/section/line triple.
|
|
834
836
|
*/
|
|
835
837
|
async function resolveFrontendPlanNewComponentSourceReferences(input) {
|
|
836
838
|
const binding = input.sourceBinding;
|
|
@@ -848,16 +850,17 @@ async function resolveFrontendPlanNewComponentSourceReferences(input) {
|
|
|
848
850
|
const fragmentsById = new Map(ledger.fragments.map((fragment) => [fragment.id, fragment]));
|
|
849
851
|
const references = new Map();
|
|
850
852
|
for (const requirement of ledger.canonicalRequirements) {
|
|
851
|
-
const
|
|
853
|
+
const citations = requirement.sourceFragmentIds
|
|
852
854
|
.map((fragmentId) => fragmentsById.get(fragmentId))
|
|
853
|
-
.
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
references.set(requirement.id, {
|
|
855
|
+
.filter((fragment) => fragment !== undefined)
|
|
856
|
+
.map((fragment) => ({
|
|
857
|
+
fragmentId: fragment.id,
|
|
857
858
|
path: fragment.path,
|
|
858
859
|
section: fragment.headingPath,
|
|
859
860
|
line: fragment.lineRange.start,
|
|
860
|
-
});
|
|
861
|
+
}));
|
|
862
|
+
if (citations.length > 0)
|
|
863
|
+
references.set(requirement.id, citations);
|
|
861
864
|
}
|
|
862
865
|
return references;
|
|
863
866
|
}
|
|
@@ -877,11 +880,27 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
877
880
|
import("typebox"),
|
|
878
881
|
import("@earendil-works/pi-coding-agent"),
|
|
879
882
|
]);
|
|
880
|
-
const { readCommittedEvents, writeTypedEventStoreJsonl } = await import("../workflows/dag/frontend-typed-event-store.js");
|
|
883
|
+
const { loadTypedEventStore, readCommittedEvents, writeTypedEventStoreJsonl, } = await import("../workflows/dag/frontend-typed-event-store.js");
|
|
881
884
|
const { adoptStagedFact, adoptTypedEventFact, stageTypedEventFact } = await import("../workflows/dag/frontend-typed-event-transaction.js");
|
|
882
885
|
const { assemblePlanPatchFromCommittedFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
|
|
883
886
|
const store = input.store;
|
|
884
887
|
const attemptId = input.attemptId;
|
|
888
|
+
// A retry creates a fresh executor-local store, but the plan ledger is the
|
|
889
|
+
// cross-attempt authority. Restore the committed prefix before registering
|
|
890
|
+
// tools; otherwise the first flush of a retry can overwrite facts that the
|
|
891
|
+
// previous attempt had already committed. The on-disk file contains only
|
|
892
|
+
// committed records, so loading it is also fail-closed with respect to
|
|
893
|
+
// staged/quarantined facts.
|
|
894
|
+
const persisted = await loadTypedEventStore(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"));
|
|
895
|
+
if (persisted.records.length > 0) {
|
|
896
|
+
const existingEventIds = new Set(store.records.map((record) => record.eventId));
|
|
897
|
+
for (const record of persisted.records) {
|
|
898
|
+
if (!existingEventIds.has(record.eventId)) {
|
|
899
|
+
store.records.push(record);
|
|
900
|
+
}
|
|
901
|
+
}
|
|
902
|
+
store.revision = Math.max(store.revision, persisted.revision);
|
|
903
|
+
}
|
|
885
904
|
const stringArray = Type.Array(Type.String({}));
|
|
886
905
|
const optionalString = Type.Optional(Type.String({}));
|
|
887
906
|
const optionalStringArray = Type.Optional(stringArray);
|
|
@@ -924,7 +943,12 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
924
943
|
consumer: optionalString,
|
|
925
944
|
}, { additionalProperties: false });
|
|
926
945
|
const mockApiSchema = Type.Object({
|
|
927
|
-
strategy: Type.
|
|
946
|
+
strategy: Type.Union([
|
|
947
|
+
Type.Literal("native"),
|
|
948
|
+
Type.Literal("browser-intercept"),
|
|
949
|
+
Type.Literal("request-adapter"),
|
|
950
|
+
Type.Literal("not-needed"),
|
|
951
|
+
], {
|
|
928
952
|
description: "native | browser-intercept | request-adapter | not-needed",
|
|
929
953
|
}),
|
|
930
954
|
activation: Type.String({}),
|
|
@@ -935,11 +959,20 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
935
959
|
paths: stringArray,
|
|
936
960
|
conflicts: stringArray,
|
|
937
961
|
}, { additionalProperties: false });
|
|
962
|
+
// Enum fields use literal unions, not advisory strings: a soft Type.String
|
|
963
|
+
// lets the model commit values like type="behavior" that pass the tool
|
|
964
|
+
// boundary, flush into the ledger, and only fail the compile-time zod enum
|
|
965
|
+
// — a deterministic attempt failure the model could have fixed in-node.
|
|
966
|
+
const verificationTargetTypeSchema = Type.Union([
|
|
967
|
+
Type.Literal("static"),
|
|
968
|
+
Type.Literal("unit"),
|
|
969
|
+
Type.Literal("component"),
|
|
970
|
+
Type.Literal("integration"),
|
|
971
|
+
Type.Literal("mock"),
|
|
972
|
+
]);
|
|
938
973
|
const verificationTargetSchema = Type.Object({
|
|
939
974
|
id: Type.String({}),
|
|
940
|
-
type:
|
|
941
|
-
description: "static | unit | component | integration | mock",
|
|
942
|
-
}),
|
|
975
|
+
type: verificationTargetTypeSchema,
|
|
943
976
|
commandLabel: Type.String({}),
|
|
944
977
|
file: Type.String({}),
|
|
945
978
|
symbol: Type.Optional(Type.String({
|
|
@@ -958,7 +991,11 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
958
991
|
const uiComponentChoiceSchema = Type.Object({
|
|
959
992
|
purpose: Type.String({}),
|
|
960
993
|
component: Type.String({}),
|
|
961
|
-
decision: Type.
|
|
994
|
+
decision: Type.Union([
|
|
995
|
+
Type.Literal("specified"),
|
|
996
|
+
Type.Literal("reuse-existing"),
|
|
997
|
+
Type.Literal("new"),
|
|
998
|
+
], {
|
|
962
999
|
description: "specified | reuse-existing | new",
|
|
963
1000
|
}),
|
|
964
1001
|
specReference: Type.Optional(Type.Object({
|
|
@@ -1033,7 +1070,7 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1033
1070
|
const recordRouteSelectionTool = defineTool({
|
|
1034
1071
|
name: "record_route_selection",
|
|
1035
1072
|
label: "record_route_selection",
|
|
1036
|
-
description: "Record the route selection needed by this plan. Repository target surface and file ownership belong to Scout/runtime.",
|
|
1073
|
+
description: "Record the route selection needed by this plan. Repository target surface and file ownership belong to Scout/runtime. Example: {\"routes\": [\"/<route>\"]}",
|
|
1037
1074
|
promptSnippet: "Record the selected routes.",
|
|
1038
1075
|
parameters: Type.Object({ routes: stringArray }, { additionalProperties: false }),
|
|
1039
1076
|
async execute(_toolCallId, params) {
|
|
@@ -1045,11 +1082,12 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1045
1082
|
const recordComponentChoiceTool = defineTool({
|
|
1046
1083
|
name: "record_component_choice",
|
|
1047
1084
|
label: "record_component_choice",
|
|
1048
|
-
description: "Record ONE component choice (origin=plan component-choice fact). Declare every UI purpose's component selection. For decision=new, pass sourceRequirementIds containing the frozen requirement ID(s) that mandate the component; the runtime derives
|
|
1049
|
-
promptSnippet: "Record
|
|
1085
|
+
description: "Record ONE component choice (origin=plan component-choice fact). Declare every UI purpose's component selection. For decision=new, pass sourceRequirementIds containing the frozen requirement ID(s) that mandate the component and sourceFragmentId selecting one frozen citation listed in the plan checklist; the runtime validates the relation and derives the exact PRD specReference. Do not read the PRD or invent a path/line. decision=reuse-existing is only for components that already exist in the repo (e.g. reusing ActiveRunBadge's styling convention). Omit rationale for reuse-existing; it is optional. Call up to 5 component choices per assistant message (batching reduces API round trips and rate-limit risk); never more than 5 per message. Optionally include stylingStrategy (set it once, on the first call). Example: {\"choice\": {\"purpose\": \"<interaction or UI state name>\", \"component\": \"<component name>\", \"decision\": \"new\"}, \"sourceRequirementIds\": [\"<AC-XXX mandating this component>\"], \"sourceFragmentId\": \"<REQ-SRC-...>\"}",
|
|
1086
|
+
promptSnippet: "Record 1-5 component choices (up to 5 per message).",
|
|
1050
1087
|
parameters: Type.Object({
|
|
1051
1088
|
choice: uiComponentChoiceSchema,
|
|
1052
1089
|
sourceRequirementIds: Type.Optional(stringArray),
|
|
1090
|
+
sourceFragmentId: optionalString,
|
|
1053
1091
|
stylingStrategy: optionalString,
|
|
1054
1092
|
}, { additionalProperties: false }),
|
|
1055
1093
|
async execute(_toolCallId, params) {
|
|
@@ -1062,6 +1100,7 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1062
1100
|
});
|
|
1063
1101
|
}
|
|
1064
1102
|
const sourceRequirementIds = stringList(params?.sourceRequirementIds);
|
|
1103
|
+
const sourceFragmentId = nonEmptyString(params?.sourceFragmentId);
|
|
1065
1104
|
const choice = { ...rawChoice };
|
|
1066
1105
|
if (choice.decision === "new") {
|
|
1067
1106
|
if (sourceRequirementIds.length === 0) {
|
|
@@ -1071,17 +1110,28 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1071
1110
|
error: "decision=new requires sourceRequirementIds so runtime can materialize the task-source specReference",
|
|
1072
1111
|
});
|
|
1073
1112
|
}
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
1113
|
+
if (!sourceFragmentId) {
|
|
1114
|
+
return planToolReceipt({
|
|
1115
|
+
ok: false,
|
|
1116
|
+
kind: "component-choice",
|
|
1117
|
+
error: "decision=new requires sourceFragmentId selecting a frozen task-source citation",
|
|
1118
|
+
});
|
|
1119
|
+
}
|
|
1120
|
+
const citation = sourceRequirementIds
|
|
1121
|
+
.flatMap((id) => input.componentNewSourceReferences?.get(id) ?? [])
|
|
1122
|
+
.find((candidate) => candidate.fragmentId === sourceFragmentId);
|
|
1123
|
+
if (!citation) {
|
|
1078
1124
|
return planToolReceipt({
|
|
1079
1125
|
ok: false,
|
|
1080
1126
|
kind: "component-choice",
|
|
1081
|
-
error: `decision=new
|
|
1127
|
+
error: `decision=new sourceFragmentId ${sourceFragmentId} is not bound to sourceRequirementIds ${sourceRequirementIds.join(", ")}`,
|
|
1082
1128
|
});
|
|
1083
1129
|
}
|
|
1084
|
-
choice.specReference =
|
|
1130
|
+
choice.specReference = {
|
|
1131
|
+
path: citation.path,
|
|
1132
|
+
section: citation.section,
|
|
1133
|
+
...(citation.line !== undefined ? { line: citation.line } : {}),
|
|
1134
|
+
};
|
|
1085
1135
|
}
|
|
1086
1136
|
const components = typeof choice.component === "string" ? [choice.component] : [];
|
|
1087
1137
|
const result = await adoptPlanFact("component-choice", `${attemptId}:record_component_choice:${randomUUID()}`, {
|
|
@@ -1093,13 +1143,22 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1093
1143
|
? { stylingStrategy: params.stylingStrategy }
|
|
1094
1144
|
: {}),
|
|
1095
1145
|
});
|
|
1096
|
-
|
|
1146
|
+
// Echo the frozen citation the runtime derived: the model sees the
|
|
1147
|
+
// purpose↔citation mapping it just committed and can re-record the
|
|
1148
|
+
// choice (last-wins per purpose at compile) when it mismatches.
|
|
1149
|
+
const echo = {
|
|
1150
|
+
...result,
|
|
1151
|
+
...(choice.specReference
|
|
1152
|
+
? { derivedSpecReference: choice.specReference }
|
|
1153
|
+
: {}),
|
|
1154
|
+
};
|
|
1155
|
+
return planToolReceipt(echo);
|
|
1097
1156
|
},
|
|
1098
1157
|
});
|
|
1099
1158
|
const recordStateFlowTool = defineTool({
|
|
1100
1159
|
name: "record_state_flow",
|
|
1101
1160
|
label: "record_state_flow",
|
|
1102
|
-
description: "Record UI states and interactions as an origin=plan state-flow fact.",
|
|
1161
|
+
description: "Record UI states and interactions as an origin=plan state-flow fact. Example: {\"uiStates\": [{\"name\": \"<state>\", \"applicable\": true, \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}], \"interactions\": [{\"name\": \"<interaction>\", \"trigger\": \"<user event>\", \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}]}",
|
|
1103
1162
|
promptSnippet: "Record the plan state-flow fact.",
|
|
1104
1163
|
parameters: Type.Object({
|
|
1105
1164
|
uiStates: Type.Array(uiStateSchema),
|
|
@@ -1177,6 +1236,26 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1177
1236
|
error: "record_state_flow requires non-empty interaction.name (id is accepted only as a legacy alias)",
|
|
1178
1237
|
});
|
|
1179
1238
|
}
|
|
1239
|
+
// Interaction -> VT forward references are legal only against
|
|
1240
|
+
// already-committed VT facts. In the segmented flow every VT
|
|
1241
|
+
// commits in the coverage segment before state flows run, so a
|
|
1242
|
+
// dangling reference here is a real defect (r20: *-BEHAVIOR
|
|
1243
|
+
// refs reached the final review untraceable).
|
|
1244
|
+
const interactionVtIds = stringList(interaction.verificationTargetIds);
|
|
1245
|
+
const committedVtIds = new Set(readCommittedEvents(store, attemptId)
|
|
1246
|
+
.map((event) => event.fact)
|
|
1247
|
+
.filter((fact) => fact.kind === "plan-verification-target")
|
|
1248
|
+
.map((fact) => fact.entry
|
|
1249
|
+
?.id)
|
|
1250
|
+
.filter((id) => typeof id === "string"));
|
|
1251
|
+
const unknownVtIds = interactionVtIds.filter((id) => !committedVtIds.has(id));
|
|
1252
|
+
if (unknownVtIds.length > 0) {
|
|
1253
|
+
return planToolReceipt({
|
|
1254
|
+
ok: false,
|
|
1255
|
+
kind: "state-flow",
|
|
1256
|
+
error: `record_state_flow interaction "${resolvedName}" references verification targets that are not recorded yet: ${unknownVtIds.join(", ")}; record them with record_plan_verification_target first, then re-record this state flow`,
|
|
1257
|
+
});
|
|
1258
|
+
}
|
|
1180
1259
|
interactions.push({ ...interaction, name: resolvedName });
|
|
1181
1260
|
}
|
|
1182
1261
|
const states = uiStates
|
|
@@ -1195,7 +1274,7 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1195
1274
|
const recordDataFlowTool = defineTool({
|
|
1196
1275
|
name: "record_data_flow",
|
|
1197
1276
|
label: "record_data_flow",
|
|
1198
|
-
description: "Record interaction/endpoint data flow as an origin=plan data-flow fact.",
|
|
1277
|
+
description: "Record interaction/endpoint data flow as an origin=plan data-flow fact. Example: {\"interactions\": [\"<interaction name>\"], \"endpoints\": [\"GET <path>\"]}",
|
|
1199
1278
|
promptSnippet: "Record the plan data-flow fact.",
|
|
1200
1279
|
parameters: Type.Object({ interactions: stringArray, endpoints: stringArray }, { additionalProperties: false }),
|
|
1201
1280
|
async execute(_toolCallId, params) {
|
|
@@ -1211,7 +1290,7 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1211
1290
|
const recordMockApiTool = defineTool({
|
|
1212
1291
|
name: "record_mock_api",
|
|
1213
1292
|
label: "record_mock_api",
|
|
1214
|
-
description: "Record the Mock/API strategy as an origin=plan mock-api fact.",
|
|
1293
|
+
description: "Record the Mock/API strategy as an origin=plan mock-api fact. Example: {\"mockApi\": {\"strategy\": \"not-needed\", \"activation\": \"n/a\", \"endpoints\": []}}",
|
|
1215
1294
|
promptSnippet: "Record the plan mock-api fact.",
|
|
1216
1295
|
parameters: Type.Object({ mockApi: mockApiSchema }, { additionalProperties: false }),
|
|
1217
1296
|
async execute(_toolCallId, params) {
|
|
@@ -1234,7 +1313,7 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1234
1313
|
const recordDesignDeviationTool = defineTool({
|
|
1235
1314
|
name: "record_design_deviation",
|
|
1236
1315
|
label: "record_design_deviation",
|
|
1237
|
-
description: "Record design evidence conflicts as an origin=plan design-deviation fact.",
|
|
1316
|
+
description: "Record design evidence conflicts as an origin=plan design-deviation fact. Example: {\"designEvidence\": {\"source\": \"<source>\", \"paths\": [\"<file>\"], \"conflicts\": [\"<conflicting requirement id>\"]}}",
|
|
1238
1317
|
promptSnippet: "Record the plan design-deviation fact.",
|
|
1239
1318
|
parameters: Type.Object({ designEvidence: designEvidenceSchema }, { additionalProperties: false }),
|
|
1240
1319
|
async execute(_toolCallId, params) {
|
|
@@ -1246,7 +1325,7 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1246
1325
|
const recordDependencyTool = defineTool({
|
|
1247
1326
|
name: "record_dependency",
|
|
1248
1327
|
label: "record_dependency",
|
|
1249
|
-
description: "Record the dependency policy as an origin=plan dependency fact.",
|
|
1328
|
+
description: "Record the dependency policy as an origin=plan dependency fact. Example: {\"policy\": \"<dependency policy statement>\"}",
|
|
1250
1329
|
promptSnippet: "Record the plan dependency fact.",
|
|
1251
1330
|
parameters: Type.Object({ policy: Type.String({}) }, { additionalProperties: false }),
|
|
1252
1331
|
async execute(_toolCallId, params) {
|
|
@@ -1266,8 +1345,8 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1266
1345
|
const recordPlanRequirementTool = defineTool({
|
|
1267
1346
|
name: "record_plan_requirement",
|
|
1268
1347
|
label: "record_plan_requirement",
|
|
1269
|
-
description: "Commit one plan requirement entry (origin=plan plan-requirement fact). Call once per requirement; entry carries id, implementationTargets, verificationTargetIds, and optional expectedOutcome (omit it — the runtime derives the outcome text from the contract requirement). Each requirement id must be recorded EXACTLY once — re-recording the same id is rejected as a duplicate and would compile a duplicated requirements[] entry. IMPORTANT:
|
|
1270
|
-
promptSnippet: "Commit
|
|
1348
|
+
description: "Commit one plan requirement entry (origin=plan plan-requirement fact). Call once per requirement; entry carries id, implementationTargets, verificationTargetIds, and optional expectedOutcome (omit it — the runtime derives the outcome text from the contract requirement). Each requirement id must be recorded EXACTLY once — re-recording the same id is rejected as a duplicate and would compile a duplicated requirements[] entry. IMPORTANT: batch up to 5 record_* calls per assistant message (4-5 entries per message minimizes API round trips and rate-limit risk); never batch more than 5 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"<AC-XXX>\", \"implementationTargets\": [\"<deliverable file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}}",
|
|
1349
|
+
promptSnippet: "Commit 1-5 plan requirement entries (up to 5 per message).",
|
|
1271
1350
|
parameters: Type.Object({
|
|
1272
1351
|
entry: requirementSchema,
|
|
1273
1352
|
}, { additionalProperties: false }),
|
|
@@ -1280,12 +1359,45 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1280
1359
|
error: "record_plan_requirement requires a non-empty entry object",
|
|
1281
1360
|
});
|
|
1282
1361
|
}
|
|
1283
|
-
|
|
1362
|
+
let entry = rawEntry;
|
|
1363
|
+
// An embedded evidenceGap with a blank description means "no gap":
|
|
1364
|
+
// small-output models emit the slot defensively with description ""
|
|
1365
|
+
// on every requirement. The canonical contract schema requires a
|
|
1366
|
+
// non-empty gap description (min 1 char), so passing the empty slot
|
|
1367
|
+
// through would deterministically fail the plan compile with
|
|
1368
|
+
// invalid-output and burn every retry. Drop the empty slot — the
|
|
1369
|
+
// field is optional and the runtime derives real blocking gaps when
|
|
1370
|
+
// a requirement has no proof.
|
|
1371
|
+
const rawGap = isRecordObject(entry.evidenceGap)
|
|
1372
|
+
? entry.evidenceGap
|
|
1373
|
+
: undefined;
|
|
1374
|
+
if (rawGap &&
|
|
1375
|
+
typeof rawGap.description === "string" &&
|
|
1376
|
+
rawGap.description.trim() === "") {
|
|
1377
|
+
const { evidenceGap: _omittedGap, ...rest } = entry;
|
|
1378
|
+
void _omittedGap;
|
|
1379
|
+
entry = rest;
|
|
1380
|
+
}
|
|
1381
|
+
// Canonical-identity check: a requirement id outside the frozen
|
|
1382
|
+
// canonical list (e.g. a BR-* business rule picked up from the PRD
|
|
1383
|
+
// prose) would commit an immutable fact that finalize's
|
|
1384
|
+
// canonical-coverage gate rejects with no in-node cure. Reject here
|
|
1385
|
+
// and name the allowed ids.
|
|
1386
|
+
const id = typeof entry.id === "string" ? entry.id : "";
|
|
1387
|
+
if (id &&
|
|
1388
|
+
input.requirementIds &&
|
|
1389
|
+
input.requirementIds.length > 0 &&
|
|
1390
|
+
!input.requirementIds.includes(id)) {
|
|
1391
|
+
return planToolReceipt({
|
|
1392
|
+
ok: false,
|
|
1393
|
+
kind: "plan-requirement",
|
|
1394
|
+
error: `record_plan_requirement id "${id}" is not a frozen canonical requirement; canonical ids are: ${input.requirementIds.join(", ")}`,
|
|
1395
|
+
});
|
|
1396
|
+
}
|
|
1284
1397
|
// A requirement id is a canonical identity: recording it twice would
|
|
1285
1398
|
// compile a duplicate requirements[] entry and fail design review.
|
|
1286
1399
|
// Reject duplicates at the tool boundary so the model can fix them
|
|
1287
1400
|
// in-node instead of burning the attempt on a later validation error.
|
|
1288
|
-
const id = typeof entry.id === "string" ? entry.id : "";
|
|
1289
1401
|
if (id) {
|
|
1290
1402
|
const existing = readCommittedEvents(store, attemptId).find((event) => {
|
|
1291
1403
|
const fact = event.fact;
|
|
@@ -1310,8 +1422,8 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1310
1422
|
const recordPlanVerificationTargetTool = defineTool({
|
|
1311
1423
|
name: "record_plan_verification_target",
|
|
1312
1424
|
label: "record_plan_verification_target",
|
|
1313
|
-
description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Call once per target; entry carries id, type, commandLabel, file, requirementIds, uiStates, and optional symbol (omit it — the trace gate verifies the file and command, not a symbol). IMPORTANT:
|
|
1314
|
-
promptSnippet: "Commit
|
|
1425
|
+
description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Call once per target; entry carries id, type, commandLabel, file, requirementIds, uiStates, and optional symbol (omit it — the trace gate verifies the file and command, not a symbol). IMPORTANT: batch up to 5 record_* calls per assistant message (4-5 entries per message minimizes API round trips and rate-limit risk); never batch more than 5 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"<VT-XXX>\", \"type\": \"unit\", \"commandLabel\": \"<frozen command label>\", \"file\": \"<test file>\", \"requirementIds\": [\"<AC-XXX>\"], \"uiStates\": []}}",
|
|
1426
|
+
promptSnippet: "Commit 1-5 plan verification target entries (up to 5 per message).",
|
|
1315
1427
|
parameters: Type.Object({
|
|
1316
1428
|
entry: verificationTargetSchema,
|
|
1317
1429
|
}, { additionalProperties: false }),
|
|
@@ -1324,6 +1436,64 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1324
1436
|
error: "record_plan_verification_target requires a non-empty entry object",
|
|
1325
1437
|
});
|
|
1326
1438
|
}
|
|
1439
|
+
// Belt-and-braces for providers that do not strictly enforce the
|
|
1440
|
+
// tool-schema enum: reject an invalid type here with the allowed
|
|
1441
|
+
// values so the model can re-record in-node instead of the whole
|
|
1442
|
+
// attempt dying at compile time on the strict zod enum.
|
|
1443
|
+
const verificationTargetType = rawEntry.type;
|
|
1444
|
+
if (typeof verificationTargetType !== "string" ||
|
|
1445
|
+
!["static", "unit", "component", "integration", "mock"].includes(verificationTargetType)) {
|
|
1446
|
+
return planToolReceipt({
|
|
1447
|
+
ok: false,
|
|
1448
|
+
kind: "plan-verification-target",
|
|
1449
|
+
error: `record_plan_verification_target entry.type must be one of static | unit | component | integration | mock (received ${JSON.stringify(verificationTargetType ?? null)}); for a runtime behavior check use type=mock or type=integration`,
|
|
1450
|
+
});
|
|
1451
|
+
}
|
|
1452
|
+
// Duplicate-id rejection: committed typed facts are immutable, so
|
|
1453
|
+
// re-recording the same VT id would deadlock the compile with a
|
|
1454
|
+
// duplicate-id error the model cannot fix in-node. Reject here so
|
|
1455
|
+
// the model submits the correction under a fresh id.
|
|
1456
|
+
const vtId = typeof rawEntry.id === "string" ? rawEntry.id : "";
|
|
1457
|
+
if (vtId &&
|
|
1458
|
+
readCommittedEvents(store, attemptId).some((event) => {
|
|
1459
|
+
const fact = event.fact;
|
|
1460
|
+
if (!fact || fact.kind !== "plan-verification-target")
|
|
1461
|
+
return false;
|
|
1462
|
+
const entryFact = fact.entry;
|
|
1463
|
+
return entryFact?.id === vtId;
|
|
1464
|
+
})) {
|
|
1465
|
+
return planToolReceipt({
|
|
1466
|
+
ok: false,
|
|
1467
|
+
kind: "plan-verification-target",
|
|
1468
|
+
error: `record_plan_verification_target duplicate: verification target ${vtId} is already recorded; submit the corrected target under a new id instead`,
|
|
1469
|
+
});
|
|
1470
|
+
}
|
|
1471
|
+
// WriteSet containment at the boundary: a committed VT fact whose
|
|
1472
|
+
// file is outside the task writeSet is immutable, and the finalize
|
|
1473
|
+
// pre-validation would then fail the whole attempt with no in-node
|
|
1474
|
+
// cure (r17 post-merge). Reject here with the allowed patterns.
|
|
1475
|
+
if (typeof rawEntry.file === "string" &&
|
|
1476
|
+
input.writeSetPatterns &&
|
|
1477
|
+
input.writeSetPatterns.length > 0 &&
|
|
1478
|
+
!input.writeSetPatterns.some((pattern) => pathMatchesPattern(rawEntry.file, pattern))) {
|
|
1479
|
+
return planToolReceipt({
|
|
1480
|
+
ok: false,
|
|
1481
|
+
kind: "plan-verification-target",
|
|
1482
|
+
error: `record_plan_verification_target file is outside the writeSet patterns [${input.writeSetPatterns.join(", ")}]: ${rawEntry.file}; verification targets must point inside the task writeSet`,
|
|
1483
|
+
});
|
|
1484
|
+
}
|
|
1485
|
+
// Symbol shape check at the boundary: a fabricated symbol committed
|
|
1486
|
+
// here is immutable (duplicate ids are rejected), and while the
|
|
1487
|
+
// compile drops it, catching it now lets the model fix the target
|
|
1488
|
+
// in one receipt-free step.
|
|
1489
|
+
if (typeof rawEntry.symbol === "string" &&
|
|
1490
|
+
isSuspiciousVerificationSymbol(rawEntry.symbol)) {
|
|
1491
|
+
return planToolReceipt({
|
|
1492
|
+
ok: false,
|
|
1493
|
+
kind: "plan-verification-target",
|
|
1494
|
+
error: `record_plan_verification_target symbol "${rawEntry.symbol}" looks fabricated; use a real exported/describe/it symbol from ${rawEntry.file ?? "the target file"} or omit the symbol entirely (the trace gate verifies file+command)`,
|
|
1495
|
+
});
|
|
1496
|
+
}
|
|
1327
1497
|
// uiStates: [] means this verification target is intentionally not
|
|
1328
1498
|
// bound to a named UI state. Keep that canonical representation even
|
|
1329
1499
|
// when a model omits the optional tool-boundary field.
|
|
@@ -1331,6 +1501,48 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1331
1501
|
...rawEntry,
|
|
1332
1502
|
uiStates: stringList(rawEntry.uiStates),
|
|
1333
1503
|
};
|
|
1504
|
+
// Cross-reference integrity at the boundary: the compile gate
|
|
1505
|
+
// rejects verification targets referencing UI states or
|
|
1506
|
+
// requirements that were never declared. Validate against the
|
|
1507
|
+
// facts already committed in this attempt so the model fixes the
|
|
1508
|
+
// reference in-node instead of burning the attempt at compile time
|
|
1509
|
+
// (r7: one full attempt lost to a single unknown UI state name).
|
|
1510
|
+
const committedEvents = readCommittedEvents(store, attemptId);
|
|
1511
|
+
const declaredUiStateNames = new Set(committedEvents.flatMap((event) => {
|
|
1512
|
+
const fact = event.fact;
|
|
1513
|
+
if (!fact || fact.kind !== "state-flow")
|
|
1514
|
+
return [];
|
|
1515
|
+
return (Array.isArray(fact.uiStates) ? fact.uiStates : [])
|
|
1516
|
+
.map((state) => isRecordObject(state) && typeof state.name === "string"
|
|
1517
|
+
? state.name
|
|
1518
|
+
: "")
|
|
1519
|
+
.filter(Boolean);
|
|
1520
|
+
}));
|
|
1521
|
+
const unknownUiStates = entry.uiStates.filter((name) => !declaredUiStateNames.has(name));
|
|
1522
|
+
if (unknownUiStates.length > 0) {
|
|
1523
|
+
return planToolReceipt({
|
|
1524
|
+
ok: false,
|
|
1525
|
+
kind: "plan-verification-target",
|
|
1526
|
+
error: `record_plan_verification_target references UI states that were never declared: ${unknownUiStates.join(", ")}; declare every referenced UI state with record_state_flow first, or pass uiStates: [] for intentionally unbound targets`,
|
|
1527
|
+
});
|
|
1528
|
+
}
|
|
1529
|
+
const declaredRequirementIds = new Set(committedEvents.flatMap((event) => {
|
|
1530
|
+
const fact = event.fact;
|
|
1531
|
+
if (!fact || fact.kind !== "plan-requirement")
|
|
1532
|
+
return [];
|
|
1533
|
+
const requirementEntry = fact.entry;
|
|
1534
|
+
return typeof requirementEntry?.id === "string"
|
|
1535
|
+
? [requirementEntry.id]
|
|
1536
|
+
: [];
|
|
1537
|
+
}));
|
|
1538
|
+
const unknownRequirementIds = stringList(rawEntry.requirementIds).filter((id) => !declaredRequirementIds.has(id));
|
|
1539
|
+
if (unknownRequirementIds.length > 0) {
|
|
1540
|
+
return planToolReceipt({
|
|
1541
|
+
ok: false,
|
|
1542
|
+
kind: "plan-verification-target",
|
|
1543
|
+
error: `record_plan_verification_target references unknown requirement ids: ${unknownRequirementIds.join(", ")}; record every referenced requirement with record_plan_requirement first`,
|
|
1544
|
+
});
|
|
1545
|
+
}
|
|
1334
1546
|
const result = await adoptPlanFact("plan-verification-target", `${attemptId}:record_plan_verification_target:${randomUUID()}`, { kind: "plan-verification-target", origin: "plan", entry });
|
|
1335
1547
|
return planToolReceipt(result);
|
|
1336
1548
|
},
|
|
@@ -1338,8 +1550,8 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1338
1550
|
const recordPlanEvidenceGapTool = defineTool({
|
|
1339
1551
|
name: "record_plan_evidence_gap",
|
|
1340
1552
|
label: "record_plan_evidence_gap",
|
|
1341
|
-
description: "Commit one plan evidence gap entry (origin=plan plan-evidence-gap fact). Call once per gap; entry carries requirementId, description, blocking. IMPORTANT:
|
|
1342
|
-
promptSnippet: "Commit
|
|
1553
|
+
description: "Commit one plan evidence gap entry (origin=plan plan-evidence-gap fact). Call once per gap; entry carries requirementId, description, blocking. IMPORTANT: batch up to 5 record_* calls per assistant message (4-5 entries per message minimizes API round trips and rate-limit risk); never batch more than 5 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"requirementId\": \"<AC-XXX>\", \"description\": \"<what evidence is missing and why>\", \"blocking\": false}}",
|
|
1554
|
+
promptSnippet: "Commit 1-5 plan evidence gap entries (up to 5 per message).",
|
|
1343
1555
|
parameters: Type.Object({
|
|
1344
1556
|
entry: evidenceGapSchema,
|
|
1345
1557
|
}, { additionalProperties: false }),
|
|
@@ -1353,6 +1565,18 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1353
1565
|
});
|
|
1354
1566
|
}
|
|
1355
1567
|
const entry = rawEntry;
|
|
1568
|
+
// A standalone evidence gap IS the gap statement: a blank description
|
|
1569
|
+
// would fail the canonical contract schema (min 1 char) after the
|
|
1570
|
+
// whole attempt finished. Reject at the boundary so the model writes
|
|
1571
|
+
// a real description in-node instead of burning the attempt.
|
|
1572
|
+
if (typeof entry.description === "string" &&
|
|
1573
|
+
entry.description.trim() === "") {
|
|
1574
|
+
return planToolReceipt({
|
|
1575
|
+
ok: false,
|
|
1576
|
+
kind: "plan-evidence-gap",
|
|
1577
|
+
error: "record_plan_evidence_gap requires a non-empty description describing the gap",
|
|
1578
|
+
});
|
|
1579
|
+
}
|
|
1356
1580
|
const result = await adoptPlanFact("plan-evidence-gap", `${attemptId}:record_plan_evidence_gap:${randomUUID()}`, { kind: "plan-evidence-gap", origin: "plan", entry });
|
|
1357
1581
|
return planToolReceipt(result);
|
|
1358
1582
|
},
|
|
@@ -1385,6 +1609,66 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1385
1609
|
? { realIntegrationGap: params.realIntegrationGap }
|
|
1386
1610
|
: {}),
|
|
1387
1611
|
};
|
|
1612
|
+
// Front-load the node's compile + policy gates into the finalize
|
|
1613
|
+
// receipt (same pipeline the design-policy shell and the node
|
|
1614
|
+
// self-check run: merge the patch onto the runtime skeleton,
|
|
1615
|
+
// then the full analyze). A failing gate used to burn an entire
|
|
1616
|
+
// attempt per finding (r8/r9: ui-design-coverage, verification
|
|
1617
|
+
// targets, UI-state shape, one attempt each); surfaced here the
|
|
1618
|
+
// model fixes the facts and re-calls finalize_plan in-node.
|
|
1619
|
+
if (input.skeleton && input.sourceBinding) {
|
|
1620
|
+
// Front-load the exact pipeline the design-policy shell and
|
|
1621
|
+
// the node self-check run (patch ⊕ skeleton -> analyze ->
|
|
1622
|
+
// policy pre-checks) into the finalize receipt. Findings
|
|
1623
|
+
// come back as fixable receipt errors instead of burning
|
|
1624
|
+
// an attempt per gate (r8/r9: coverage, verification
|
|
1625
|
+
// targets, UI-state shape each cost a full attempt).
|
|
1626
|
+
const { analyzeFrontendPlanPatchCandidate, applyFrontendContractMergePatch, FrontendContractFailure, PlanPolicyPrecheckFailure, serializeDeterministicJson, } = await import("../workflows/dag/frontend-implementation-contract.js");
|
|
1627
|
+
try {
|
|
1628
|
+
const merged = applyFrontendContractMergePatch(input.skeleton, patch);
|
|
1629
|
+
await analyzeFrontendPlanPatchCandidate({
|
|
1630
|
+
runDir: input.runDir,
|
|
1631
|
+
rawContractText: serializeDeterministicJson(merged),
|
|
1632
|
+
sourceBinding: input.sourceBinding,
|
|
1633
|
+
});
|
|
1634
|
+
}
|
|
1635
|
+
catch (error) {
|
|
1636
|
+
if (error instanceof PlanPolicyPrecheckFailure) {
|
|
1637
|
+
// Template the fix: every uncovered interaction / state
|
|
1638
|
+
// maps to a ready-to-submit record_component_choice
|
|
1639
|
+
// call. One reuse-existing choice covers all
|
|
1640
|
+
// behavioural interactions.
|
|
1641
|
+
const suggestions = error.findings
|
|
1642
|
+
.filter((finding) => finding.code === "ui-design-coverage-missing" &&
|
|
1643
|
+
finding.path)
|
|
1644
|
+
.map((finding) => ({
|
|
1645
|
+
tool: "record_component_choice",
|
|
1646
|
+
args: {
|
|
1647
|
+
choice: {
|
|
1648
|
+
purpose: finding.path,
|
|
1649
|
+
component: "<name the existing or new component>",
|
|
1650
|
+
decision: "reuse-existing",
|
|
1651
|
+
},
|
|
1652
|
+
},
|
|
1653
|
+
}));
|
|
1654
|
+
const suggestionBlock = suggestions.length > 0
|
|
1655
|
+
? ` Suggested record_* calls (copy, fill component, submit): ${JSON.stringify(suggestions)}`
|
|
1656
|
+
: "";
|
|
1657
|
+
return planToolReceipt({
|
|
1658
|
+
ok: false,
|
|
1659
|
+
kind: "finalize_plan",
|
|
1660
|
+
error: `finalize_plan pre-validation failed (fix the listed plan facts with record_* tools, then call finalize_plan again): ${error.message}${suggestionBlock}`,
|
|
1661
|
+
});
|
|
1662
|
+
}
|
|
1663
|
+
if (!(error instanceof FrontendContractFailure))
|
|
1664
|
+
throw error;
|
|
1665
|
+
return planToolReceipt({
|
|
1666
|
+
ok: false,
|
|
1667
|
+
kind: "finalize_plan",
|
|
1668
|
+
error: `finalize_plan pre-validation failed (fix the listed plan facts with record_* tools, then call finalize_plan again): ${error.message}`,
|
|
1669
|
+
});
|
|
1670
|
+
}
|
|
1671
|
+
}
|
|
1388
1672
|
const patchResult = await adoptPlanFact("target-surface", `${attemptId}:finalize_plan:patch:${randomUUID()}`, { kind: "target-surface", origin: "plan", patch });
|
|
1389
1673
|
if (!patchResult.ok) {
|
|
1390
1674
|
return planToolReceipt({
|
|
@@ -1464,6 +1748,19 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1464
1748
|
const committed = readCommittedEvents(store, attemptId);
|
|
1465
1749
|
await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"), committed);
|
|
1466
1750
|
},
|
|
1751
|
+
committedFactCount: () => readCommittedEvents(store, attemptId).length,
|
|
1752
|
+
committedRequirementIds: () => {
|
|
1753
|
+
const ids = new Set();
|
|
1754
|
+
for (const event of readCommittedEvents(store, attemptId)) {
|
|
1755
|
+
const fact = event.fact;
|
|
1756
|
+
if (!fact || fact.kind !== "plan-requirement")
|
|
1757
|
+
continue;
|
|
1758
|
+
const entryFact = fact.entry;
|
|
1759
|
+
if (typeof entryFact?.id === "string")
|
|
1760
|
+
ids.add(entryFact.id);
|
|
1761
|
+
}
|
|
1762
|
+
return ids;
|
|
1763
|
+
},
|
|
1467
1764
|
};
|
|
1468
1765
|
}
|
|
1469
1766
|
function isRecordObject(value) {
|
|
@@ -1597,8 +1894,8 @@ export async function createFrontendContractTools(input) {
|
|
|
1597
1894
|
const recordTools = Object.entries(recordKinds).map(([name, kind]) => defineTool({
|
|
1598
1895
|
name,
|
|
1599
1896
|
label: name,
|
|
1600
|
-
description: `Commit an origin=contract ${kind} fact.`,
|
|
1601
|
-
promptSnippet: `Commit
|
|
1897
|
+
description: `Commit an origin=contract ${kind} fact. IMPORTANT: submit incrementally — batch up to 5 record_* calls per message, starting from the FIRST message; never attempt to emit the whole contract in one response (a single large dump will be truncated and rejected). Every message must make progress by committing at least one record_* fact.`,
|
|
1898
|
+
promptSnippet: `Commit 1-5 origin=contract ${kind} facts (up to 5 per message).`,
|
|
1602
1899
|
parameters: Type.Object({}, { additionalProperties: true }),
|
|
1603
1900
|
async execute(_toolCallId, params) {
|
|
1604
1901
|
const result = await adoptContractFact(kind, {
|
|
@@ -1753,6 +2050,22 @@ export async function createFrontendScoutEvidenceTools(input) {
|
|
|
1753
2050
|
continue;
|
|
1754
2051
|
}
|
|
1755
2052
|
try {
|
|
2053
|
+
// Directories are legitimate named targets (greenfield smoke: the
|
|
2054
|
+
// page directory exists while the files inside it are to be
|
|
2055
|
+
// created). readFile on a directory throws EISDIR, which used to
|
|
2056
|
+
// mark every directory path fresh=false and structurally fail the
|
|
2057
|
+
// freshness gate for create-new surfaces. stat() first: a directory
|
|
2058
|
+
// counts as fresh existence evidence; its content hash is a stable
|
|
2059
|
+
// directory marker since there is no single file content to hash.
|
|
2060
|
+
const info = await stat(absolute);
|
|
2061
|
+
if (info.isDirectory()) {
|
|
2062
|
+
evidence.push({
|
|
2063
|
+
path: relative,
|
|
2064
|
+
sha256: createHash("sha256").update(`directory:${relative}`).digest("hex"),
|
|
2065
|
+
fresh: true,
|
|
2066
|
+
});
|
|
2067
|
+
continue;
|
|
2068
|
+
}
|
|
1756
2069
|
const bytes = await readFile(absolute);
|
|
1757
2070
|
evidence.push({
|
|
1758
2071
|
path: relative,
|
|
@@ -1803,7 +2116,7 @@ export async function createFrontendScoutEvidenceTools(input) {
|
|
|
1803
2116
|
const recordTargetSurfaceTool = defineTool({
|
|
1804
2117
|
name: "record_target_surface",
|
|
1805
2118
|
label: "record_target_surface",
|
|
1806
|
-
description: "Commit an origin=scout target-surface fact with complete/blocked discovery status. A complete surface needs a proven target path and no unresolved paths; blocked surfaces name the unresolved paths instead of guessing.",
|
|
2119
|
+
description: "Commit an origin=scout target-surface fact with complete/blocked discovery status. A complete surface needs a proven target path and no unresolved paths; blocked surfaces name the unresolved paths instead of guessing. Example: {\"completeness\": \"complete\", \"entrypoint\": \"<file>\", \"implementationPaths\": [\"<dir or file>\"], \"testPaths\": [\"<file>\"], \"allowedPathConflicts\": [], \"unresolvedPaths\": []}",
|
|
1807
2120
|
promptSnippet: "Commit an origin=scout target-surface fact.",
|
|
1808
2121
|
parameters: Type.Object({
|
|
1809
2122
|
completeness: scoutCompleteness,
|
|
@@ -1842,7 +2155,7 @@ export async function createFrontendScoutEvidenceTools(input) {
|
|
|
1842
2155
|
const recordDesignEvidenceTool = defineTool({
|
|
1843
2156
|
name: "record_design_evidence",
|
|
1844
2157
|
label: "record_design_evidence",
|
|
1845
|
-
description: "Commit an origin=scout design-evidence fact (source, paths, conflicts).",
|
|
2158
|
+
description: "Commit an origin=scout design-evidence fact (source, paths, conflicts). Example: {\"source\": \"<source>\", \"paths\": [\"<file>\"], \"conflicts\": []}",
|
|
1846
2159
|
promptSnippet: "Commit an origin=scout design-evidence fact.",
|
|
1847
2160
|
parameters: Type.Object({ source: Type.String({}), paths: stringArray, conflicts: stringArray }, { additionalProperties: false }),
|
|
1848
2161
|
async execute(_toolCallId, params) {
|
|
@@ -2239,6 +2552,171 @@ async function runFrontendDesignTerminalShadow(input) {
|
|
|
2239
2552
|
}
|
|
2240
2553
|
return input.mapped;
|
|
2241
2554
|
}
|
|
2555
|
+
/** Tool subsets for the three frontend plan sessions (r17 split design). */
|
|
2556
|
+
const FRONTEND_PLAN_SEGMENTS = [
|
|
2557
|
+
{
|
|
2558
|
+
id: "coverage",
|
|
2559
|
+
toolNames: new Set([
|
|
2560
|
+
"record_plan_requirement",
|
|
2561
|
+
"record_plan_verification_target",
|
|
2562
|
+
"adopt_staged_fact",
|
|
2563
|
+
]),
|
|
2564
|
+
instruction: [
|
|
2565
|
+
"PLAN SEGMENT 1/3 — coverage mapping only.",
|
|
2566
|
+
"Your ONLY job: for every frozen requirement, emit record_plan_requirement (requirement → implementation files) and record_plan_verification_target (verification target bound to requirement ids and files). Do NOT record components, UI states, mock, dependency, or routes — a follow-up session owns those.",
|
|
2567
|
+
"Do not call finalize_plan; it is not available in this segment.",
|
|
2568
|
+
].join(" "),
|
|
2569
|
+
},
|
|
2570
|
+
{
|
|
2571
|
+
id: "ux-decisions",
|
|
2572
|
+
toolNames: new Set([
|
|
2573
|
+
"record_component_choice",
|
|
2574
|
+
"record_state_flow",
|
|
2575
|
+
"record_data_flow",
|
|
2576
|
+
"record_mock_api",
|
|
2577
|
+
"record_design_deviation",
|
|
2578
|
+
"record_dependency",
|
|
2579
|
+
"record_route_selection",
|
|
2580
|
+
"adopt_staged_fact",
|
|
2581
|
+
]),
|
|
2582
|
+
instruction: [
|
|
2583
|
+
"PLAN SEGMENT 2/3 — UX decisions.",
|
|
2584
|
+
"Requirements and verification targets are already committed in the ledger (do NOT re-record them; duplicates are rejected). Your ONLY job: record component choices (decision=new requires sourceRequirementIds per the citation table), state flows, data flow, mock strategy, design deviation, dependency policy, and route selection.",
|
|
2585
|
+
"Do not call finalize_plan; it is not available in this segment.",
|
|
2586
|
+
].join(" "),
|
|
2587
|
+
},
|
|
2588
|
+
{
|
|
2589
|
+
id: "finalize",
|
|
2590
|
+
toolNames: null,
|
|
2591
|
+
instruction: [
|
|
2592
|
+
"PLAN SEGMENT 3/3 — finalize.",
|
|
2593
|
+
"All record_* tools are available: if a finalize_plan receipt reports missing or invalid facts, fix them with the named record_* tool and call finalize_plan again. Otherwise call finalize_plan exactly once with no extra fields.",
|
|
2594
|
+
].join(" "),
|
|
2595
|
+
},
|
|
2596
|
+
];
|
|
2597
|
+
/**
|
|
2598
|
+
* Coverage-batch slicing for the frontend plan split (options 1+5): the
|
|
2599
|
+
* coverage segment becomes one session per requirement slice (default 4
|
|
2600
|
+
* requirements), so a small output budget can never be exhausted by
|
|
2601
|
+
* upfront reasoning about the whole requirement list. Zero-progress
|
|
2602
|
+
* batches are split in half and retried (option 5); single-requirement
|
|
2603
|
+
* zero-progress failures short-circuit to the retry ladder.
|
|
2604
|
+
*/
|
|
2605
|
+
const FRONTEND_PLAN_COVERAGE_BATCH_SIZE = 4;
|
|
2606
|
+
const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 32;
|
|
2607
|
+
export async function runFrontendPlanSegmentedSessions(input) {
|
|
2608
|
+
const queue = [];
|
|
2609
|
+
// An empty list means "ledger unreadable / unknown" and must fall back to
|
|
2610
|
+
// one unscoped coverage session — only a non-empty list batches.
|
|
2611
|
+
const requirementIdsProvided = input.requirementIds !== undefined && input.requirementIds.length > 0;
|
|
2612
|
+
const pending = (input.requirementIds ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
|
|
2613
|
+
for (const segment of FRONTEND_PLAN_SEGMENTS) {
|
|
2614
|
+
if (segment.id !== "coverage") {
|
|
2615
|
+
queue.push({
|
|
2616
|
+
id: segment.id,
|
|
2617
|
+
toolNames: segment.toolNames,
|
|
2618
|
+
prompt: `${input.basePrompt}\n\n${segment.instruction}`,
|
|
2619
|
+
});
|
|
2620
|
+
continue;
|
|
2621
|
+
}
|
|
2622
|
+
// No requirement list (unreadable ledger) -> one unscoped coverage
|
|
2623
|
+
// session. A provided list with everything committed (resume) skips
|
|
2624
|
+
// coverage entirely.
|
|
2625
|
+
if (!requirementIdsProvided || pending.length > 0) {
|
|
2626
|
+
if (!requirementIdsProvided) {
|
|
2627
|
+
queue.push({
|
|
2628
|
+
id: segment.id,
|
|
2629
|
+
toolNames: segment.toolNames,
|
|
2630
|
+
prompt: `${input.basePrompt}\n\n${segment.instruction}`,
|
|
2631
|
+
});
|
|
2632
|
+
continue;
|
|
2633
|
+
}
|
|
2634
|
+
for (let i = 0; i < pending.length; i += FRONTEND_PLAN_COVERAGE_BATCH_SIZE) {
|
|
2635
|
+
const slice = pending.slice(i, i + FRONTEND_PLAN_COVERAGE_BATCH_SIZE);
|
|
2636
|
+
queue.push({
|
|
2637
|
+
id: `coverage-batch-${i / FRONTEND_PLAN_COVERAGE_BATCH_SIZE + 1}`,
|
|
2638
|
+
toolNames: segment.toolNames,
|
|
2639
|
+
coverageSlice: slice,
|
|
2640
|
+
prompt: `${input.basePrompt}\n\n${segment.instruction}\n\nCOVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them.`,
|
|
2641
|
+
});
|
|
2642
|
+
}
|
|
2643
|
+
}
|
|
2644
|
+
}
|
|
2645
|
+
let last;
|
|
2646
|
+
let index = 0;
|
|
2647
|
+
while (index < queue.length && index < FRONTEND_PLAN_BATCH_MAX_SESSIONS) {
|
|
2648
|
+
const session = queue[index];
|
|
2649
|
+
const remaining = (session.coverageSlice ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
|
|
2650
|
+
// Resume/earlier-batch commits may already cover this slice.
|
|
2651
|
+
if (session.coverageSlice && remaining.length === 0) {
|
|
2652
|
+
index += 1;
|
|
2653
|
+
continue;
|
|
2654
|
+
}
|
|
2655
|
+
let prompt = session.prompt;
|
|
2656
|
+
if (session.coverageSlice) {
|
|
2657
|
+
prompt = prompt.replace(/COVERAGE BATCH: process ONLY these requirements in this session: .*/, `COVERAGE BATCH: process ONLY these requirements in this session: ${remaining.join(", ")}. Other requirements are handled by separate sessions; do not record them.`);
|
|
2658
|
+
}
|
|
2659
|
+
const committedBefore = input.committedFactCount();
|
|
2660
|
+
const customTools = input.segmentCustomTools(session.toolNames);
|
|
2661
|
+
const result = await input.piStepFn({
|
|
2662
|
+
...input.sessionOptions,
|
|
2663
|
+
prompt,
|
|
2664
|
+
...(customTools.length > 0
|
|
2665
|
+
? {
|
|
2666
|
+
writerToolPolicy: {
|
|
2667
|
+
requireSdk: true,
|
|
2668
|
+
customTools,
|
|
2669
|
+
},
|
|
2670
|
+
}
|
|
2671
|
+
: {}),
|
|
2672
|
+
});
|
|
2673
|
+
last = result;
|
|
2674
|
+
try {
|
|
2675
|
+
await input.flushLedger();
|
|
2676
|
+
}
|
|
2677
|
+
catch {
|
|
2678
|
+
// best-effort: the node-level flush runs again after the attempt
|
|
2679
|
+
}
|
|
2680
|
+
const committedAfter = input.committedFactCount();
|
|
2681
|
+
if (result.ok) {
|
|
2682
|
+
index += 1;
|
|
2683
|
+
continue;
|
|
2684
|
+
}
|
|
2685
|
+
const committedFactsOnlySuccess = session.id !== "finalize" &&
|
|
2686
|
+
!(result.assistantText ?? "").trim() &&
|
|
2687
|
+
!result.stderr.trim() &&
|
|
2688
|
+
!result.timedOut &&
|
|
2689
|
+
committedAfter > committedBefore;
|
|
2690
|
+
if (committedFactsOnlySuccess) {
|
|
2691
|
+
// The session died but banked facts: keep the progress and move on.
|
|
2692
|
+
index += 1;
|
|
2693
|
+
continue;
|
|
2694
|
+
}
|
|
2695
|
+
// Option 5: a multi-requirement coverage batch that failed with ZERO
|
|
2696
|
+
// new facts and no provider stderr is the upfront-reasoning burn —
|
|
2697
|
+
// halve the slice and retry instead of failing the attempt.
|
|
2698
|
+
const coverageSlice = session.coverageSlice;
|
|
2699
|
+
const zeroProgressBurn = coverageSlice !== undefined &&
|
|
2700
|
+
coverageSlice.length > 1 &&
|
|
2701
|
+
committedAfter === committedBefore &&
|
|
2702
|
+
!(result.assistantText ?? "").trim() &&
|
|
2703
|
+
!result.stderr.trim() &&
|
|
2704
|
+
!result.timedOut;
|
|
2705
|
+
if (zeroProgressBurn && coverageSlice) {
|
|
2706
|
+
const half = Math.ceil(coverageSlice.length / 2);
|
|
2707
|
+
queue.splice(index, 1, { ...session, coverageSlice: coverageSlice.slice(0, half) }, { ...session, coverageSlice: coverageSlice.slice(half) });
|
|
2708
|
+
continue;
|
|
2709
|
+
}
|
|
2710
|
+
return result;
|
|
2711
|
+
}
|
|
2712
|
+
return (last ?? {
|
|
2713
|
+
ok: false,
|
|
2714
|
+
stdout: "",
|
|
2715
|
+
stderr: "frontend plan segmentation produced no session",
|
|
2716
|
+
failureCategory: "empty-output",
|
|
2717
|
+
durationMs: 0,
|
|
2718
|
+
});
|
|
2719
|
+
}
|
|
2242
2720
|
export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
|
|
2243
2721
|
const started = Date.now();
|
|
2244
2722
|
const persona = resolveDagPiPersona(input.task);
|
|
@@ -2524,6 +3002,8 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
2524
3002
|
runDir: meta.runDir,
|
|
2525
3003
|
nodeId: input.task.id,
|
|
2526
3004
|
skeleton: input.task.structuredContractOutput?.skeleton,
|
|
3005
|
+
sourceBinding: meta.spec.sourceBinding,
|
|
3006
|
+
writeSetPatterns: input.task.writeSet,
|
|
2527
3007
|
componentNewSourceReferences: await resolveFrontendPlanNewComponentSourceReferences({
|
|
2528
3008
|
cwd: input.cwd,
|
|
2529
3009
|
sourceBinding: meta.spec.sourceBinding,
|
|
@@ -2619,10 +3099,9 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
2619
3099
|
const piExtensionPaths = piExtensionsResolution && resolvePiBackend() !== "cli-only"
|
|
2620
3100
|
? piExtensionsResolution.resolved.flatMap((entry) => entry.entryPaths)
|
|
2621
3101
|
: undefined;
|
|
2622
|
-
|
|
3102
|
+
const piSessionOptions = {
|
|
2623
3103
|
attachedFiles: [],
|
|
2624
3104
|
modelConfig: resolveDagPiModelConfig(input.model, input.thinking ? { thinking: input.thinking } : undefined),
|
|
2625
|
-
prompt: input.prompt,
|
|
2626
3105
|
repoRoot: input.cwd,
|
|
2627
3106
|
step,
|
|
2628
3107
|
toolNames: resolveDagPiToolNames(input.task),
|
|
@@ -2635,7 +3114,6 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
2635
3114
|
...(input.task.contextBudget
|
|
2636
3115
|
? { contextBudget: input.task.contextBudget }
|
|
2637
3116
|
: {}),
|
|
2638
|
-
...(writerToolPolicy ? { writerToolPolicy } : {}),
|
|
2639
3117
|
...(piExtensionPaths && piExtensionPaths.length > 0
|
|
2640
3118
|
? { piExtensionPaths }
|
|
2641
3119
|
: {}),
|
|
@@ -2648,7 +3126,55 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
2648
3126
|
bridgeActivity(activity.kind, activity.at);
|
|
2649
3127
|
}
|
|
2650
3128
|
: undefined,
|
|
2651
|
-
}
|
|
3129
|
+
};
|
|
3130
|
+
if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
|
|
3131
|
+
// Frontend-only split: three sequential sessions with independent
|
|
3132
|
+
// output budgets (coverage -> UX decisions -> finalize), mirroring the
|
|
3133
|
+
// backend-test template's module sharding. Every other template keeps
|
|
3134
|
+
// the single-session path below.
|
|
3135
|
+
let planRequirementIds = [];
|
|
3136
|
+
try {
|
|
3137
|
+
const { readCommittedOriginFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
|
|
3138
|
+
const contractFacts = await readCommittedOriginFacts(meta.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl");
|
|
3139
|
+
// Only requirement facts: the contract ledger also carries
|
|
3140
|
+
// constraints (CON-*), evidence expectations (EV-*), handoff
|
|
3141
|
+
// intents (HND-*), open questions (OQ-*) and split proposals
|
|
3142
|
+
// (SPLIT-*) that all have ids — feeding those into the coverage
|
|
3143
|
+
// batches made the model record non-frozen plan-requirement ids
|
|
3144
|
+
// that finalize's canonical-coverage gate then rejected (r-ext2).
|
|
3145
|
+
planRequirementIds = contractFacts
|
|
3146
|
+
.filter((record) => record.fact
|
|
3147
|
+
?.kind === "requirement")
|
|
3148
|
+
.map((record) => record.fact?.id)
|
|
3149
|
+
.filter((id) => typeof id === "string")
|
|
3150
|
+
.sort();
|
|
3151
|
+
}
|
|
3152
|
+
catch {
|
|
3153
|
+
// Unreadable ledger falls back to a single coverage session.
|
|
3154
|
+
}
|
|
3155
|
+
result = await runFrontendPlanSegmentedSessions({
|
|
3156
|
+
piStepFn,
|
|
3157
|
+
sessionOptions: piSessionOptions,
|
|
3158
|
+
basePrompt: input.prompt,
|
|
3159
|
+
attempt: input.attempt ?? 1,
|
|
3160
|
+
committedFactCount: () => planLedgerTools.committedFactCount(),
|
|
3161
|
+
requirementIds: planRequirementIds,
|
|
3162
|
+
committedRequirementIds: () => planLedgerTools.committedRequirementIds(),
|
|
3163
|
+
segmentCustomTools: (toolNames) => toolNames === null
|
|
3164
|
+
? planLedgerTools.customTools
|
|
3165
|
+
: planLedgerTools.customTools.filter((tool) => typeof tool === "object" &&
|
|
3166
|
+
tool !== null &&
|
|
3167
|
+
toolNames.has(tool.name)),
|
|
3168
|
+
flushLedger: () => planLedgerTools.flush(),
|
|
3169
|
+
});
|
|
3170
|
+
}
|
|
3171
|
+
else {
|
|
3172
|
+
result = await piStepFn({
|
|
3173
|
+
...piSessionOptions,
|
|
3174
|
+
prompt: input.prompt,
|
|
3175
|
+
...(writerToolPolicy ? { writerToolPolicy } : {}),
|
|
3176
|
+
});
|
|
3177
|
+
}
|
|
2652
3178
|
}
|
|
2653
3179
|
catch (error) {
|
|
2654
3180
|
if (playwrightToolContext) {
|
|
@@ -2785,27 +3311,28 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
2785
3311
|
catch {
|
|
2786
3312
|
// best-effort flush; missing ledger still fails at the node validator
|
|
2787
3313
|
}
|
|
2788
|
-
//
|
|
2789
|
-
//
|
|
2790
|
-
//
|
|
2791
|
-
//
|
|
2792
|
-
// session log and retry with a reduced-reading instruction.
|
|
2793
|
-
const readBurstIssues = input.task.readBudget?.onExhaustion === "return-guidance"
|
|
2794
|
-
? []
|
|
2795
|
-
: await detectPlanReadBurst({
|
|
2796
|
-
runDir: meta.runDir,
|
|
2797
|
-
nodeId: input.task.id,
|
|
2798
|
-
});
|
|
2799
|
-
if (readBurstIssues.length > 0) {
|
|
2800
|
-
return {
|
|
2801
|
-
...mapped,
|
|
2802
|
-
ok: false,
|
|
2803
|
-
failureCategory: "read-burst",
|
|
2804
|
-
stderr: [mapped.stderr, ...readBurstIssues].filter(Boolean).join("\n\n"),
|
|
2805
|
-
};
|
|
2806
|
-
}
|
|
3314
|
+
// NOTE: the legacy plan read-burst guard lived here. It is dead code
|
|
3315
|
+
// since the plan node went tool-only (resolveDagPiToolNames grants no
|
|
3316
|
+
// read/grep/ls/find), so a read burst is structurally impossible; the
|
|
3317
|
+
// read-budget path (scout etc.) keeps its own guard.
|
|
2807
3318
|
}
|
|
2808
3319
|
if (!isWriteTask) {
|
|
3320
|
+
if (!mapped.ok &&
|
|
3321
|
+
!(mapped.assistantText ?? "").trim() &&
|
|
3322
|
+
!mapped.stderr.trim()) {
|
|
3323
|
+
// Terminal-fact acceptance: the run finished with every fact committed
|
|
3324
|
+
// (including the terminal) but no final narrative text. A timeout or
|
|
3325
|
+
// provider error always leaves supervision/provider stderr, so a
|
|
3326
|
+
// blank stderr here means the only "failure" is the empty text.
|
|
3327
|
+
const terminalAccepted = await acceptCommittedTypedTerminalFact(meta.runDir, input.task.id);
|
|
3328
|
+
if (terminalAccepted) {
|
|
3329
|
+
return {
|
|
3330
|
+
...mapped,
|
|
3331
|
+
ok: true,
|
|
3332
|
+
failureCategory: undefined,
|
|
3333
|
+
};
|
|
3334
|
+
}
|
|
3335
|
+
}
|
|
2809
3336
|
return mapped;
|
|
2810
3337
|
}
|
|
2811
3338
|
let writeGuardOk = true;
|