@tea-agent/loop-agent 0.39.0-beta.12 → 0.39.0-beta.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/CHANGELOG.md +27 -0
  2. package/dist/application/dag/generate-task-dag.js +6 -2
  3. package/dist/application/task-lifecycle/advance.js +20 -4
  4. package/dist/application/task-lifecycle/observe.js +171 -17
  5. package/dist/application/task-lifecycle/plan-transitions.js +42 -7
  6. package/dist/build-stamp.json +3 -3
  7. package/dist/commands/client-recovery.js +8 -36
  8. package/dist/executors/dag-pi-executor.js +636 -109
  9. package/dist/executors/pi-executor.js +8 -4
  10. package/dist/executors/pi-sdk-executor.js +33 -5
  11. package/dist/executors/shell-executor.js +102 -32
  12. package/dist/governance/checks.js +1 -0
  13. package/dist/shared/pi-context-pressure/checkpoint.js +116 -0
  14. package/dist/shared/pi-context-pressure/compaction-policy.js +151 -0
  15. package/dist/shared/pi-context-pressure/env.js +58 -0
  16. package/dist/shared/pi-context-pressure/extension.js +100 -0
  17. package/dist/shared/pi-context-pressure/index.js +7 -0
  18. package/dist/shared/pi-context-pressure/overflow.js +252 -0
  19. package/dist/shared/pi-context-pressure/sift-bridge.js +386 -0
  20. package/dist/shared/pi-context-pressure/telemetry.js +51 -0
  21. package/dist/task/frontend-project-capability.js +3 -1
  22. package/dist/task/source-prepare/fragment-inventory.js +4 -1
  23. package/dist/worker/console/chat/pi-runtime.js +146 -5
  24. package/dist/worker/console/chat/provider-error.js +2 -1
  25. package/dist/worker/console/chat/routes.js +3 -0
  26. package/dist/worker/console/chat/sift-bridge.js +1 -0
  27. package/dist/worker/console/dag-execution-receipt.js +20 -2
  28. package/dist/worker/console/operator-actions.js +4 -3
  29. package/dist/worker/console/static/assets/{abnfDiagram-N423BO3Z-CXj_GnSb.js → abnfDiagram-N423BO3Z-DC863mud.js} +1 -1
  30. package/dist/worker/console/static/assets/{arc-BZp6JAp7.js → arc-CftC38G9.js} +1 -1
  31. package/dist/worker/console/static/assets/{architectureDiagram-T3A2C74G-DfpcEuYU.js → architectureDiagram-T3A2C74G-B3PAPQCw.js} +1 -1
  32. package/dist/worker/console/static/assets/{blockDiagram-VBNYF7ZC-_Gf0xadb.js → blockDiagram-VBNYF7ZC-C9UDJNRv.js} +1 -1
  33. package/dist/worker/console/static/assets/{c4Diagram-5PPSVZJV-Ct2QPCmv.js → c4Diagram-5PPSVZJV--BJ76fv_.js} +1 -1
  34. package/dist/worker/console/static/assets/channel-CUz-Bg86.js +1 -0
  35. package/dist/worker/console/static/assets/{chunk-2GRJ4B5K-BfklkzKl.js → chunk-2GRJ4B5K--hiIqoGp.js} +1 -1
  36. package/dist/worker/console/static/assets/{chunk-2Q5K7J3B-Dy25vZJV.js → chunk-2Q5K7J3B-DgTzpIa3.js} +1 -1
  37. package/dist/worker/console/static/assets/{chunk-5RXB4S5H-BFlCRZep.js → chunk-5RXB4S5H-DIwpJziP.js} +1 -1
  38. package/dist/worker/console/static/assets/{chunk-5VM5RSS4-op3oVxIE.js → chunk-5VM5RSS4-BJzbdUi0.js} +1 -1
  39. package/dist/worker/console/static/assets/{chunk-6Q2QTUOP-C0R7rzF2.js → chunk-6Q2QTUOP-BbGouI1Z.js} +1 -1
  40. package/dist/worker/console/static/assets/{chunk-GF5L2VYU-Bt44TCGy.js → chunk-GF5L2VYU-B9247w69.js} +1 -1
  41. package/dist/worker/console/static/assets/{chunk-JWPE2WC7-_uEE_XFx.js → chunk-JWPE2WC7-CXob_wYy.js} +1 -1
  42. package/dist/worker/console/static/assets/{chunk-KBJHAD2P-C3TOYGZ9.js → chunk-KBJHAD2P-C6FUel1g.js} +1 -1
  43. package/dist/worker/console/static/assets/{chunk-RYQCIY6F-Cz60oBPV.js → chunk-RYQCIY6F-DGKRYKHu.js} +1 -1
  44. package/dist/worker/console/static/assets/{chunk-XXDRQBXY-8Bik0qis.js → chunk-XXDRQBXY-kNBqNPfx.js} +1 -1
  45. package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-BhlUacmD.js +1 -0
  46. package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-BhlUacmD.js +1 -0
  47. package/dist/worker/console/static/assets/{cose-bilkent-JH36ORCC-C2QIOA4a.js → cose-bilkent-JH36ORCC--xmkDwfD.js} +1 -1
  48. package/dist/worker/console/static/assets/{cynefin-VYW2F7L2-CSH_yUUd.js → cynefin-VYW2F7L2-Cw5FMMuG.js} +1 -1
  49. package/dist/worker/console/static/assets/{cynefinDiagram-MW4NZA55-BDLw2XFM.js → cynefinDiagram-MW4NZA55-CuLslt2t.js} +1 -1
  50. package/dist/worker/console/static/assets/{dagre-VZM6K2ZE-CcD9ZtF4.js → dagre-VZM6K2ZE-D9ngqrs2.js} +1 -1
  51. package/dist/worker/console/static/assets/{diagram-7IWD3JNH-DqwtkBsI.js → diagram-7IWD3JNH-RDRSbHQp.js} +1 -1
  52. package/dist/worker/console/static/assets/{diagram-B4RE2ZJO-BWVEcqsC.js → diagram-B4RE2ZJO-BgQ1P9aV.js} +1 -1
  53. package/dist/worker/console/static/assets/{diagram-LBJQPF4R-M-WAbtQi.js → diagram-LBJQPF4R-D3WWyPtn.js} +1 -1
  54. package/dist/worker/console/static/assets/{diagram-Q27KOJAE-DKW7SixP.js → diagram-Q27KOJAE-L2k28wTR.js} +1 -1
  55. package/dist/worker/console/static/assets/{diagram-UB23O5K3-PHHPPtrj.js → diagram-UB23O5K3-DDNrA5Qw.js} +1 -1
  56. package/dist/worker/console/static/assets/{ebnfDiagram-BXEA7PRR-BQD-B4RX.js → ebnfDiagram-BXEA7PRR-BaRFCOdM.js} +1 -1
  57. package/dist/worker/console/static/assets/{erDiagram-JOGREHBK-BFvULb51.js → erDiagram-JOGREHBK-CM33DVVM.js} +1 -1
  58. package/dist/worker/console/static/assets/{flowDiagram-UKHOOZJN-Bw-aXTCb.js → flowDiagram-UKHOOZJN-CVBSjOxe.js} +1 -1
  59. package/dist/worker/console/static/assets/{ganttDiagram-PKOTCBZU-BGKw04Qx.js → ganttDiagram-PKOTCBZU-BdscewE_.js} +1 -1
  60. package/dist/worker/console/static/assets/{gitGraphDiagram-DS77QQ5N-RqKaPR-I.js → gitGraphDiagram-DS77QQ5N-HRuSZb7g.js} +1 -1
  61. package/dist/worker/console/static/assets/{index-B_D8rbWc.js → index-B9JJQsVK.js} +102 -72
  62. package/dist/worker/console/static/assets/index-rWaGv4jz.css +1 -0
  63. package/dist/worker/console/static/assets/{infoDiagram-6WML65LV-YO5dnzrJ.js → infoDiagram-6WML65LV-cYfHvfAR.js} +1 -1
  64. package/dist/worker/console/static/assets/{ishikawaDiagram-WSZJBQD7-Bi1VZHMq.js → ishikawaDiagram-WSZJBQD7-CZmaoOuy.js} +1 -1
  65. package/dist/worker/console/static/assets/{journeyDiagram-NVQOT4AX-Bszls1DD.js → journeyDiagram-NVQOT4AX-ntYijh7g.js} +1 -1
  66. package/dist/worker/console/static/assets/{kanban-definition-27J2QSJJ-PhzeaZ09.js → kanban-definition-27J2QSJJ-Cz3b0DeD.js} +1 -1
  67. package/dist/worker/console/static/assets/{linear-BsjbDoXi.js → linear-DvGonpsP.js} +1 -1
  68. package/dist/worker/console/static/assets/{mermaid.core-0B7NnWKk.js → mermaid.core-B-3vjfyW.js} +5 -5
  69. package/dist/worker/console/static/assets/{mindmap-definition-FAOFIHXS-BJr4Fj-q.js → mindmap-definition-FAOFIHXS-5iKxlsX3.js} +1 -1
  70. package/dist/worker/console/static/assets/{pegDiagram-VL7TDLO6-moC4fpGB.js → pegDiagram-VL7TDLO6-C8BUwapW.js} +1 -1
  71. package/dist/worker/console/static/assets/{pieDiagram-7S7Q4E2Y-BEw37-2c.js → pieDiagram-7S7Q4E2Y-B2rPoQQG.js} +1 -1
  72. package/dist/worker/console/static/assets/{quadrantDiagram-CIZ2JOQS-Cq6LyasU.js → quadrantDiagram-CIZ2JOQS-jDeVUAy4.js} +1 -1
  73. package/dist/worker/console/static/assets/{railroadDiagram-AXF67PYL-DUCMcK0D.js → railroadDiagram-AXF67PYL-DV140Dwm.js} +1 -1
  74. package/dist/worker/console/static/assets/{requirementDiagram-LRYGKXZP-C3upTZm7.js → requirementDiagram-LRYGKXZP-CnYgHtYC.js} +1 -1
  75. package/dist/worker/console/static/assets/{sankeyDiagram-W5VNT64P-BI_gMsCW.js → sankeyDiagram-W5VNT64P-SIK3CdWw.js} +1 -1
  76. package/dist/worker/console/static/assets/{sequenceDiagram-SI44F4Z6-YFOIRzfN.js → sequenceDiagram-SI44F4Z6-DW_vZix7.js} +1 -1
  77. package/dist/worker/console/static/assets/{sizeCapture-X5ZJPWSS-dOnB7UDD.js → sizeCapture-X5ZJPWSS-NBAtkg0C.js} +1 -1
  78. package/dist/worker/console/static/assets/{stateDiagram-OKZ733FA-BXUniaIh.js → stateDiagram-OKZ733FA-Bi1bQxpi.js} +1 -1
  79. package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-DjWKYjQ1.js +1 -0
  80. package/dist/worker/console/static/assets/{swimlanes-SLNWSIFB-2tA4wTNu.js → swimlanes-SLNWSIFB-8PT_uP_i.js} +2 -2
  81. package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-B4cdm7bH.js +8 -0
  82. package/dist/worker/console/static/assets/{timeline-definition-Z64GVDOM-DO0HkJXC.js → timeline-definition-Z64GVDOM-CD0ZZPk0.js} +1 -1
  83. package/dist/worker/console/static/assets/{vennDiagram-T6HMQDX7-DvIOzixv.js → vennDiagram-T6HMQDX7-BbEgWdhK.js} +1 -1
  84. package/dist/worker/console/static/assets/{wardleyDiagram-T6FBY63Y-DXDZ0cTj.js → wardleyDiagram-T6FBY63Y-DCwPKdLl.js} +1 -1
  85. package/dist/worker/console/static/assets/{xychartDiagram-ELKLHX3M-B32Ark0D.js → xychartDiagram-ELKLHX3M-sfC5PcNg.js} +1 -1
  86. package/dist/worker/console/static/index.html +2 -2
  87. package/dist/worker/observe/static/styles.css +9 -0
  88. package/dist/worker/observe/static/views/dag-inspector.js +40 -0
  89. package/dist/worker/observe/static/views/session-timeline.js +135 -0
  90. package/dist/workflows/dag/backend-test-case-coverage-analysis.js +157 -7
  91. package/dist/workflows/dag/backend-test-pytest-collection.js +70 -2
  92. package/dist/workflows/dag/backend-test-result-contract.js +4 -0
  93. package/dist/workflows/dag/backend-test-scenario-param.js +339 -53
  94. package/dist/workflows/dag/backend-test-writer-completeness.js +11 -0
  95. package/dist/workflows/dag/dag-retry-schema.js +138 -0
  96. package/dist/workflows/dag/frontend-implementation-contract.js +174 -31
  97. package/dist/workflows/dag/frontend-review-context.js +12 -1
  98. package/dist/workflows/dag/frontend-shadow-dual-write.js +59 -13
  99. package/dist/workflows/dag/frontend-writer-admission.js +13 -0
  100. package/dist/workflows/dag/init-hybrid.js +8 -3
  101. package/dist/workflows/dag/node-execution.js +277 -3
  102. package/dist/workflows/dag/rerun-feedback.js +135 -3
  103. package/dist/workflows/dag/retry-policy.js +13 -122
  104. package/dist/workflows/dag/types.js +11 -4
  105. package/docs/architecture/runtime-boundaries.md +2 -1
  106. package/docs/templates/backend-test-dag.json +4 -3
  107. package/harness.json +3 -3
  108. package/package.json +4 -3
  109. package/dist/worker/console/static/assets/channel-3TxJgYaH.js +0 -1
  110. package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-BHIkXpp3.js +0 -1
  111. package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-BHIkXpp3.js +0 -1
  112. package/dist/worker/console/static/assets/index-BdNx6fj0.css +0 -1
  113. package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-Cdi6UhLa.js +0 -1
  114. package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-D-RJBbb0.js +0 -8
@@ -1,6 +1,6 @@
1
1
  import path from "node:path";
2
2
  import { createHash, randomUUID } from "node:crypto";
3
- import { readFile } from "node:fs/promises";
3
+ import { readFile, stat } from "node:fs/promises";
4
4
  import { writeDagNodeJsonArtifact, writeTextArtifactFile, } from "../infrastructure/harness/artifact-store.js";
5
5
  import { writeJsonAtomic } from "../infrastructure/harness/atomic-write.js";
6
6
  import { mapContractBlockedOwner } from "../workflows/dag/frontend-human-decision.js";
@@ -15,6 +15,8 @@ import { parseLedgerJson } from "../task/source-prepare/ledger.js";
15
15
  import { redactPromptForLog, truncateOutput, } from "../shared/output-truncation.js";
16
16
  import { GitStatusUnavailableError, pathsChangedDuringRun, readGitStatusPorcelain, recoverRootNulArtifact, snapshotGitStatusPathFingerprints, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
17
17
  import { captureWorkspaceWriteSnapshot, diffWorkspaceWriteSnapshots, } from "./workspace-write-snapshot.js";
18
+ import { pathMatchesPattern } from "../shared/git-progress.js";
19
+ import { isSuspiciousVerificationSymbol } from "../workflows/dag/frontend-implementation-contract.js";
18
20
  import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
19
21
  import { assessBackendTestMdPlanCompleteness, assessBackendTestMdWriterCompleteness, assessBackendTestPytestPlanCompleteness, assessBackendTestPytestWriterCompleteness, assessBackendTestShardChildCompleteness, backendTestWriterProgressRoleForTask, classifyBackendTestWriterCompletenessFailure, isBackendTestCompletenessRetryCandidate, isBackendTestMdPlanTask, isBackendTestPytestPlanTask, isBackendTestShardChildTask, writeBackendTestWriterProgressArtifacts, } from "../workflows/dag/backend-test-writer-completeness.js";
20
22
  import { resolveBackendTestLayout } from "../workflows/dag/backend-test-layout.js";
@@ -261,6 +263,45 @@ export function isFrontendScoutEvidenceNode(task) {
261
263
  export function isFrontendPlanLedgerNode(task) {
262
264
  return task.id === "frontend-plan-pi";
263
265
  }
266
+ /** Facts-first nodes whose authoritative output is a committed typed terminal
267
+ * fact (not the assistant text). Downstream compilation reads the flushed
268
+ * `<nodeId>/<file>` store and never the node narrative, so a committed
269
+ * terminal means the work is done. */
270
+ const TYPED_TERMINAL_FACT_NODES = {
271
+ "frontend-contract-pi": {
272
+ file: "contract-typed-facts.jsonl",
273
+ kind: "contract-finalized",
274
+ },
275
+ "frontend-plan-pi": {
276
+ file: "plan-typed-facts.jsonl",
277
+ kind: "finalize_plan",
278
+ },
279
+ };
280
+ /**
281
+ * Accept a facts-terminal node result whose final assistant text is blank
282
+ * when the typed terminal fact was committed successfully. Small-output
283
+ * models legitimately end after the terminal tool call; without this the
284
+ * empty assistantText fails the node as empty-output, the failure classifier
285
+ * phrase-scans the whole session stream and can mislabel the committed run
286
+ * as rate-limit/network, and the finished ledger is thrown away for a
287
+ * deterministic retry that burns the full prompt budget again. Fail-closed:
288
+ * acceptance requires a committed terminal record from the flushed typed
289
+ * facts store; provider-error attempts (non-empty stderr) are never accepted
290
+ * by the caller.
291
+ */
292
+ export async function acceptCommittedTypedTerminalFact(runDir, nodeId) {
293
+ const binding = TYPED_TERMINAL_FACT_NODES[nodeId];
294
+ if (!binding)
295
+ return false;
296
+ try {
297
+ const { readCommittedOriginFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
298
+ const records = await readCommittedOriginFacts(runDir, nodeId, binding.file);
299
+ return records.some((record) => record.fact.kind === binding.kind);
300
+ }
301
+ catch {
302
+ return false;
303
+ }
304
+ }
264
305
  export function resolveDagPiToolNames(task) {
265
306
  if (isFrontendReviewTypedTerminalNode(task)) {
266
307
  return [
@@ -277,8 +318,11 @@ export function resolveDagPiToolNames(task) {
277
318
  ];
278
319
  }
279
320
  if (isFrontendContractTypedNode(task)) {
321
+ // Contract is an incremental-commit node: the source-fidelity ledger is
322
+ // compiled into the <frontend_contract_input> block (node-execution), so
323
+ // no read tools — mirrors the plan node. Omitting read tools prevents a
324
+ // contract from spending its output budget re-reading the raw source.
280
325
  return [
281
- ...DAG_PI_READONLY_TOOLS,
282
326
  ...FRONTEND_CONTRACT_RECORD_TOOL_NAMES,
283
327
  ...FRONTEND_CONTRACT_TERMINAL_TOOL_NAMES,
284
328
  ];
@@ -342,10 +386,6 @@ export function scanReviewTerminalKindsFromSessionEvents(content) {
342
386
  }
343
387
  return kinds;
344
388
  }
345
- /** Read-only discovery tool budget for frontend-plan-pi. Exceeding it means
346
- * the plan re-read upstream outputs/sources instead of trusting typed facts,
347
- * which blows up the context window (400 request-too-large). */
348
- const PLAN_READ_TOOL_BUDGET = 40;
349
389
  const READ_ONLY_TOOL_NAMES = new Set(["read", "grep", "ls", "find"]);
350
390
  function readEventPath(event) {
351
391
  const candidates = [event.path, event.readPath, event.input];
@@ -430,45 +470,6 @@ export async function detectNodeReadBudget(input) {
430
470
  issues.push(`frontend ${input.nodeId} read budget exceeded: ${stats.elapsedMs}ms (budget ${input.budget.maxMs}ms)`);
431
471
  return issues;
432
472
  }
433
- /** Deterministic read-burst guard for frontend-plan-pi: count read-only
434
- * discovery tool calls (read/grep/ls/find) from the session log. Over budget →
435
- * read-burst, retried with a reduced-reading instruction. Pure scan; a
436
- * successful plan under budget is never blocked. */
437
- export async function detectPlanReadBurst(input) {
438
- const sessionEventsPath = path.join(input.runDir, input.nodeId, "session-events.jsonl");
439
- let count = 0;
440
- try {
441
- const content = await readFile(sessionEventsPath, "utf8");
442
- for (const line of content.split("\n")) {
443
- if (!line.trim())
444
- continue;
445
- try {
446
- const event = JSON.parse(line);
447
- if (event.type === "tool_execution_start" &&
448
- typeof event.toolName === "string" &&
449
- (event.toolName === "read" ||
450
- event.toolName === "grep" ||
451
- event.toolName === "ls" ||
452
- event.toolName === "find")) {
453
- count += 1;
454
- }
455
- }
456
- catch {
457
- // skip unparseable line
458
- }
459
- }
460
- }
461
- catch {
462
- // Missing/unreadable session log → no burst detection
463
- return [];
464
- }
465
- if (count > PLAN_READ_TOOL_BUDGET) {
466
- return [
467
- `frontend plan read-burst: ${count} read-only tool calls (budget ${PLAN_READ_TOOL_BUDGET}). Trust the upstream contract/scout typed facts; do not re-read contract/scout outputs or source files already captured. Minimize discovery reads, commit record_* facts directly, then finalize_plan.`,
468
- ];
469
- }
470
- return [];
471
- }
472
473
  export const FRONTEND_DESIGN_TERMINAL_TOOL_NAMES = new Set([
473
474
  "approve_design",
474
475
  "request_design_changes",
@@ -829,8 +830,9 @@ async function loadContractRequirementInheritance(runDir) {
829
830
  }
830
831
  /**
831
832
  * Resolve task-source citations from the source-fidelity ledger before the
832
- * planner starts. The planner names a frozen requirement id; it never needs
833
- * to re-read a PRD merely to recover a path/section/line triple.
833
+ * planner starts. The planner names a frozen requirement id and one of its
834
+ * fragment ids; it never needs to re-read a PRD merely to recover a
835
+ * path/section/line triple.
834
836
  */
835
837
  async function resolveFrontendPlanNewComponentSourceReferences(input) {
836
838
  const binding = input.sourceBinding;
@@ -848,16 +850,17 @@ async function resolveFrontendPlanNewComponentSourceReferences(input) {
848
850
  const fragmentsById = new Map(ledger.fragments.map((fragment) => [fragment.id, fragment]));
849
851
  const references = new Map();
850
852
  for (const requirement of ledger.canonicalRequirements) {
851
- const fragment = requirement.sourceFragmentIds
853
+ const citations = requirement.sourceFragmentIds
852
854
  .map((fragmentId) => fragmentsById.get(fragmentId))
853
- .find((candidate) => candidate !== undefined);
854
- if (!fragment)
855
- continue;
856
- references.set(requirement.id, {
855
+ .filter((fragment) => fragment !== undefined)
856
+ .map((fragment) => ({
857
+ fragmentId: fragment.id,
857
858
  path: fragment.path,
858
859
  section: fragment.headingPath,
859
860
  line: fragment.lineRange.start,
860
- });
861
+ }));
862
+ if (citations.length > 0)
863
+ references.set(requirement.id, citations);
861
864
  }
862
865
  return references;
863
866
  }
@@ -877,11 +880,27 @@ export async function createFrontendPlanLedgerTools(input) {
877
880
  import("typebox"),
878
881
  import("@earendil-works/pi-coding-agent"),
879
882
  ]);
880
- const { readCommittedEvents, writeTypedEventStoreJsonl } = await import("../workflows/dag/frontend-typed-event-store.js");
883
+ const { loadTypedEventStore, readCommittedEvents, writeTypedEventStoreJsonl, } = await import("../workflows/dag/frontend-typed-event-store.js");
881
884
  const { adoptStagedFact, adoptTypedEventFact, stageTypedEventFact } = await import("../workflows/dag/frontend-typed-event-transaction.js");
882
885
  const { assemblePlanPatchFromCommittedFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
883
886
  const store = input.store;
884
887
  const attemptId = input.attemptId;
888
+ // A retry creates a fresh executor-local store, but the plan ledger is the
889
+ // cross-attempt authority. Restore the committed prefix before registering
890
+ // tools; otherwise the first flush of a retry can overwrite facts that the
891
+ // previous attempt had already committed. The on-disk file contains only
892
+ // committed records, so loading it is also fail-closed with respect to
893
+ // staged/quarantined facts.
894
+ const persisted = await loadTypedEventStore(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"));
895
+ if (persisted.records.length > 0) {
896
+ const existingEventIds = new Set(store.records.map((record) => record.eventId));
897
+ for (const record of persisted.records) {
898
+ if (!existingEventIds.has(record.eventId)) {
899
+ store.records.push(record);
900
+ }
901
+ }
902
+ store.revision = Math.max(store.revision, persisted.revision);
903
+ }
885
904
  const stringArray = Type.Array(Type.String({}));
886
905
  const optionalString = Type.Optional(Type.String({}));
887
906
  const optionalStringArray = Type.Optional(stringArray);
@@ -924,7 +943,12 @@ export async function createFrontendPlanLedgerTools(input) {
924
943
  consumer: optionalString,
925
944
  }, { additionalProperties: false });
926
945
  const mockApiSchema = Type.Object({
927
- strategy: Type.String({
946
+ strategy: Type.Union([
947
+ Type.Literal("native"),
948
+ Type.Literal("browser-intercept"),
949
+ Type.Literal("request-adapter"),
950
+ Type.Literal("not-needed"),
951
+ ], {
928
952
  description: "native | browser-intercept | request-adapter | not-needed",
929
953
  }),
930
954
  activation: Type.String({}),
@@ -935,11 +959,20 @@ export async function createFrontendPlanLedgerTools(input) {
935
959
  paths: stringArray,
936
960
  conflicts: stringArray,
937
961
  }, { additionalProperties: false });
962
+ // Enum fields use literal unions, not advisory strings: a soft Type.String
963
+ // lets the model commit values like type="behavior" that pass the tool
964
+ // boundary, flush into the ledger, and only fail the compile-time zod enum
965
+ // — a deterministic attempt failure the model could have fixed in-node.
966
+ const verificationTargetTypeSchema = Type.Union([
967
+ Type.Literal("static"),
968
+ Type.Literal("unit"),
969
+ Type.Literal("component"),
970
+ Type.Literal("integration"),
971
+ Type.Literal("mock"),
972
+ ]);
938
973
  const verificationTargetSchema = Type.Object({
939
974
  id: Type.String({}),
940
- type: Type.String({
941
- description: "static | unit | component | integration | mock",
942
- }),
975
+ type: verificationTargetTypeSchema,
943
976
  commandLabel: Type.String({}),
944
977
  file: Type.String({}),
945
978
  symbol: Type.Optional(Type.String({
@@ -958,7 +991,11 @@ export async function createFrontendPlanLedgerTools(input) {
958
991
  const uiComponentChoiceSchema = Type.Object({
959
992
  purpose: Type.String({}),
960
993
  component: Type.String({}),
961
- decision: Type.String({
994
+ decision: Type.Union([
995
+ Type.Literal("specified"),
996
+ Type.Literal("reuse-existing"),
997
+ Type.Literal("new"),
998
+ ], {
962
999
  description: "specified | reuse-existing | new",
963
1000
  }),
964
1001
  specReference: Type.Optional(Type.Object({
@@ -1033,7 +1070,7 @@ export async function createFrontendPlanLedgerTools(input) {
1033
1070
  const recordRouteSelectionTool = defineTool({
1034
1071
  name: "record_route_selection",
1035
1072
  label: "record_route_selection",
1036
- description: "Record the route selection needed by this plan. Repository target surface and file ownership belong to Scout/runtime.",
1073
+ description: "Record the route selection needed by this plan. Repository target surface and file ownership belong to Scout/runtime. Example: {\"routes\": [\"/<route>\"]}",
1037
1074
  promptSnippet: "Record the selected routes.",
1038
1075
  parameters: Type.Object({ routes: stringArray }, { additionalProperties: false }),
1039
1076
  async execute(_toolCallId, params) {
@@ -1045,11 +1082,12 @@ export async function createFrontendPlanLedgerTools(input) {
1045
1082
  const recordComponentChoiceTool = defineTool({
1046
1083
  name: "record_component_choice",
1047
1084
  label: "record_component_choice",
1048
- description: "Record ONE component choice (origin=plan component-choice fact). Declare every UI purpose's component selection. For decision=new, pass sourceRequirementIds containing the frozen requirement ID(s) that mandate the component; the runtime derives its exact PRD specReference from the source-fidelity ledger. Do not read the PRD or invent a path/line. decision=reuse-existing is only for components that already exist in the repo (e.g. reusing ActiveRunBadge's styling convention). Omit rationale for reuse-existing; it is optional. Call once per component — one tool call per message. Optionally include stylingStrategy (set it once, on the first call).",
1049
- promptSnippet: "Record one component choice (one tool call per message).",
1085
+ description: "Record ONE component choice (origin=plan component-choice fact). Declare every UI purpose's component selection. For decision=new, pass sourceRequirementIds containing the frozen requirement ID(s) that mandate the component and sourceFragmentId selecting one frozen citation listed in the plan checklist; the runtime validates the relation and derives the exact PRD specReference. Do not read the PRD or invent a path/line. decision=reuse-existing is only for components that already exist in the repo (e.g. reusing ActiveRunBadge's styling convention). Omit rationale for reuse-existing; it is optional. Call up to 5 component choices per assistant message (batching reduces API round trips and rate-limit risk); never more than 5 per message. Optionally include stylingStrategy (set it once, on the first call). Example: {\"choice\": {\"purpose\": \"<interaction or UI state name>\", \"component\": \"<component name>\", \"decision\": \"new\"}, \"sourceRequirementIds\": [\"<AC-XXX mandating this component>\"], \"sourceFragmentId\": \"<REQ-SRC-...>\"}",
1086
+ promptSnippet: "Record 1-5 component choices (up to 5 per message).",
1050
1087
  parameters: Type.Object({
1051
1088
  choice: uiComponentChoiceSchema,
1052
1089
  sourceRequirementIds: Type.Optional(stringArray),
1090
+ sourceFragmentId: optionalString,
1053
1091
  stylingStrategy: optionalString,
1054
1092
  }, { additionalProperties: false }),
1055
1093
  async execute(_toolCallId, params) {
@@ -1062,6 +1100,7 @@ export async function createFrontendPlanLedgerTools(input) {
1062
1100
  });
1063
1101
  }
1064
1102
  const sourceRequirementIds = stringList(params?.sourceRequirementIds);
1103
+ const sourceFragmentId = nonEmptyString(params?.sourceFragmentId);
1065
1104
  const choice = { ...rawChoice };
1066
1105
  if (choice.decision === "new") {
1067
1106
  if (sourceRequirementIds.length === 0) {
@@ -1071,17 +1110,28 @@ export async function createFrontendPlanLedgerTools(input) {
1071
1110
  error: "decision=new requires sourceRequirementIds so runtime can materialize the task-source specReference",
1072
1111
  });
1073
1112
  }
1074
- const specReference = sourceRequirementIds
1075
- .map((id) => input.componentNewSourceReferences?.get(id))
1076
- .find((reference) => reference !== undefined);
1077
- if (!specReference) {
1113
+ if (!sourceFragmentId) {
1114
+ return planToolReceipt({
1115
+ ok: false,
1116
+ kind: "component-choice",
1117
+ error: "decision=new requires sourceFragmentId selecting a frozen task-source citation",
1118
+ });
1119
+ }
1120
+ const citation = sourceRequirementIds
1121
+ .flatMap((id) => input.componentNewSourceReferences?.get(id) ?? [])
1122
+ .find((candidate) => candidate.fragmentId === sourceFragmentId);
1123
+ if (!citation) {
1078
1124
  return planToolReceipt({
1079
1125
  ok: false,
1080
1126
  kind: "component-choice",
1081
- error: `decision=new sourceRequirementIds have no frozen task-source citation: ${sourceRequirementIds.join(", ")}`,
1127
+ error: `decision=new sourceFragmentId ${sourceFragmentId} is not bound to sourceRequirementIds ${sourceRequirementIds.join(", ")}`,
1082
1128
  });
1083
1129
  }
1084
- choice.specReference = specReference;
1130
+ choice.specReference = {
1131
+ path: citation.path,
1132
+ section: citation.section,
1133
+ ...(citation.line !== undefined ? { line: citation.line } : {}),
1134
+ };
1085
1135
  }
1086
1136
  const components = typeof choice.component === "string" ? [choice.component] : [];
1087
1137
  const result = await adoptPlanFact("component-choice", `${attemptId}:record_component_choice:${randomUUID()}`, {
@@ -1093,13 +1143,22 @@ export async function createFrontendPlanLedgerTools(input) {
1093
1143
  ? { stylingStrategy: params.stylingStrategy }
1094
1144
  : {}),
1095
1145
  });
1096
- return planToolReceipt(result);
1146
+ // Echo the frozen citation the runtime derived: the model sees the
1147
+ // purpose↔citation mapping it just committed and can re-record the
1148
+ // choice (last-wins per purpose at compile) when it mismatches.
1149
+ const echo = {
1150
+ ...result,
1151
+ ...(choice.specReference
1152
+ ? { derivedSpecReference: choice.specReference }
1153
+ : {}),
1154
+ };
1155
+ return planToolReceipt(echo);
1097
1156
  },
1098
1157
  });
1099
1158
  const recordStateFlowTool = defineTool({
1100
1159
  name: "record_state_flow",
1101
1160
  label: "record_state_flow",
1102
- description: "Record UI states and interactions as an origin=plan state-flow fact.",
1161
+ description: "Record UI states and interactions as an origin=plan state-flow fact. Example: {\"uiStates\": [{\"name\": \"<state>\", \"applicable\": true, \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}], \"interactions\": [{\"name\": \"<interaction>\", \"trigger\": \"<user event>\", \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}]}",
1103
1162
  promptSnippet: "Record the plan state-flow fact.",
1104
1163
  parameters: Type.Object({
1105
1164
  uiStates: Type.Array(uiStateSchema),
@@ -1177,6 +1236,26 @@ export async function createFrontendPlanLedgerTools(input) {
1177
1236
  error: "record_state_flow requires non-empty interaction.name (id is accepted only as a legacy alias)",
1178
1237
  });
1179
1238
  }
1239
+ // Interaction -> VT forward references are legal only against
1240
+ // already-committed VT facts. In the segmented flow every VT
1241
+ // commits in the coverage segment before state flows run, so a
1242
+ // dangling reference here is a real defect (r20: *-BEHAVIOR
1243
+ // refs reached the final review untraceable).
1244
+ const interactionVtIds = stringList(interaction.verificationTargetIds);
1245
+ const committedVtIds = new Set(readCommittedEvents(store, attemptId)
1246
+ .map((event) => event.fact)
1247
+ .filter((fact) => fact.kind === "plan-verification-target")
1248
+ .map((fact) => fact.entry
1249
+ ?.id)
1250
+ .filter((id) => typeof id === "string"));
1251
+ const unknownVtIds = interactionVtIds.filter((id) => !committedVtIds.has(id));
1252
+ if (unknownVtIds.length > 0) {
1253
+ return planToolReceipt({
1254
+ ok: false,
1255
+ kind: "state-flow",
1256
+ error: `record_state_flow interaction "${resolvedName}" references verification targets that are not recorded yet: ${unknownVtIds.join(", ")}; record them with record_plan_verification_target first, then re-record this state flow`,
1257
+ });
1258
+ }
1180
1259
  interactions.push({ ...interaction, name: resolvedName });
1181
1260
  }
1182
1261
  const states = uiStates
@@ -1195,7 +1274,7 @@ export async function createFrontendPlanLedgerTools(input) {
1195
1274
  const recordDataFlowTool = defineTool({
1196
1275
  name: "record_data_flow",
1197
1276
  label: "record_data_flow",
1198
- description: "Record interaction/endpoint data flow as an origin=plan data-flow fact.",
1277
+ description: "Record interaction/endpoint data flow as an origin=plan data-flow fact. Example: {\"interactions\": [\"<interaction name>\"], \"endpoints\": [\"GET <path>\"]}",
1199
1278
  promptSnippet: "Record the plan data-flow fact.",
1200
1279
  parameters: Type.Object({ interactions: stringArray, endpoints: stringArray }, { additionalProperties: false }),
1201
1280
  async execute(_toolCallId, params) {
@@ -1211,7 +1290,7 @@ export async function createFrontendPlanLedgerTools(input) {
1211
1290
  const recordMockApiTool = defineTool({
1212
1291
  name: "record_mock_api",
1213
1292
  label: "record_mock_api",
1214
- description: "Record the Mock/API strategy as an origin=plan mock-api fact.",
1293
+ description: "Record the Mock/API strategy as an origin=plan mock-api fact. Example: {\"mockApi\": {\"strategy\": \"not-needed\", \"activation\": \"n/a\", \"endpoints\": []}}",
1215
1294
  promptSnippet: "Record the plan mock-api fact.",
1216
1295
  parameters: Type.Object({ mockApi: mockApiSchema }, { additionalProperties: false }),
1217
1296
  async execute(_toolCallId, params) {
@@ -1234,7 +1313,7 @@ export async function createFrontendPlanLedgerTools(input) {
1234
1313
  const recordDesignDeviationTool = defineTool({
1235
1314
  name: "record_design_deviation",
1236
1315
  label: "record_design_deviation",
1237
- description: "Record design evidence conflicts as an origin=plan design-deviation fact.",
1316
+ description: "Record design evidence conflicts as an origin=plan design-deviation fact. Example: {\"designEvidence\": {\"source\": \"<source>\", \"paths\": [\"<file>\"], \"conflicts\": [\"<conflicting requirement id>\"]}}",
1238
1317
  promptSnippet: "Record the plan design-deviation fact.",
1239
1318
  parameters: Type.Object({ designEvidence: designEvidenceSchema }, { additionalProperties: false }),
1240
1319
  async execute(_toolCallId, params) {
@@ -1246,7 +1325,7 @@ export async function createFrontendPlanLedgerTools(input) {
1246
1325
  const recordDependencyTool = defineTool({
1247
1326
  name: "record_dependency",
1248
1327
  label: "record_dependency",
1249
- description: "Record the dependency policy as an origin=plan dependency fact.",
1328
+ description: "Record the dependency policy as an origin=plan dependency fact. Example: {\"policy\": \"<dependency policy statement>\"}",
1250
1329
  promptSnippet: "Record the plan dependency fact.",
1251
1330
  parameters: Type.Object({ policy: Type.String({}) }, { additionalProperties: false }),
1252
1331
  async execute(_toolCallId, params) {
@@ -1266,8 +1345,8 @@ export async function createFrontendPlanLedgerTools(input) {
1266
1345
  const recordPlanRequirementTool = defineTool({
1267
1346
  name: "record_plan_requirement",
1268
1347
  label: "record_plan_requirement",
1269
- description: "Commit one plan requirement entry (origin=plan plan-requirement fact). Call once per requirement; entry carries id, implementationTargets, verificationTargetIds, and optional expectedOutcome (omit it — the runtime derives the outcome text from the contract requirement). Each requirement id must be recorded EXACTLY once — re-recording the same id is rejected as a duplicate and would compile a duplicated requirements[] entry. IMPORTANT: call this tool exactly ONE time per assistant message — never batch multiple record_* calls together in one message; emit one call, wait for its result, then call the next.",
1270
- promptSnippet: "Commit one plan requirement entry (one tool call per message).",
1348
+ description: "Commit one plan requirement entry (origin=plan plan-requirement fact). Call once per requirement; entry carries id, implementationTargets, verificationTargetIds, and optional expectedOutcome (omit it — the runtime derives the outcome text from the contract requirement). Each requirement id must be recorded EXACTLY once — re-recording the same id is rejected as a duplicate and would compile a duplicated requirements[] entry. IMPORTANT: batch up to 5 record_* calls per assistant message (4-5 entries per message minimizes API round trips and rate-limit risk); never batch more than 5 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"<AC-XXX>\", \"implementationTargets\": [\"<deliverable file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}}",
1349
+ promptSnippet: "Commit 1-5 plan requirement entries (up to 5 per message).",
1271
1350
  parameters: Type.Object({
1272
1351
  entry: requirementSchema,
1273
1352
  }, { additionalProperties: false }),
@@ -1280,12 +1359,45 @@ export async function createFrontendPlanLedgerTools(input) {
1280
1359
  error: "record_plan_requirement requires a non-empty entry object",
1281
1360
  });
1282
1361
  }
1283
- const entry = rawEntry;
1362
+ let entry = rawEntry;
1363
+ // An embedded evidenceGap with a blank description means "no gap":
1364
+ // small-output models emit the slot defensively with description ""
1365
+ // on every requirement. The canonical contract schema requires a
1366
+ // non-empty gap description (min 1 char), so passing the empty slot
1367
+ // through would deterministically fail the plan compile with
1368
+ // invalid-output and burn every retry. Drop the empty slot — the
1369
+ // field is optional and the runtime derives real blocking gaps when
1370
+ // a requirement has no proof.
1371
+ const rawGap = isRecordObject(entry.evidenceGap)
1372
+ ? entry.evidenceGap
1373
+ : undefined;
1374
+ if (rawGap &&
1375
+ typeof rawGap.description === "string" &&
1376
+ rawGap.description.trim() === "") {
1377
+ const { evidenceGap: _omittedGap, ...rest } = entry;
1378
+ void _omittedGap;
1379
+ entry = rest;
1380
+ }
1381
+ // Canonical-identity check: a requirement id outside the frozen
1382
+ // canonical list (e.g. a BR-* business rule picked up from the PRD
1383
+ // prose) would commit an immutable fact that finalize's
1384
+ // canonical-coverage gate rejects with no in-node cure. Reject here
1385
+ // and name the allowed ids.
1386
+ const id = typeof entry.id === "string" ? entry.id : "";
1387
+ if (id &&
1388
+ input.requirementIds &&
1389
+ input.requirementIds.length > 0 &&
1390
+ !input.requirementIds.includes(id)) {
1391
+ return planToolReceipt({
1392
+ ok: false,
1393
+ kind: "plan-requirement",
1394
+ error: `record_plan_requirement id "${id}" is not a frozen canonical requirement; canonical ids are: ${input.requirementIds.join(", ")}`,
1395
+ });
1396
+ }
1284
1397
  // A requirement id is a canonical identity: recording it twice would
1285
1398
  // compile a duplicate requirements[] entry and fail design review.
1286
1399
  // Reject duplicates at the tool boundary so the model can fix them
1287
1400
  // in-node instead of burning the attempt on a later validation error.
1288
- const id = typeof entry.id === "string" ? entry.id : "";
1289
1401
  if (id) {
1290
1402
  const existing = readCommittedEvents(store, attemptId).find((event) => {
1291
1403
  const fact = event.fact;
@@ -1310,8 +1422,8 @@ export async function createFrontendPlanLedgerTools(input) {
1310
1422
  const recordPlanVerificationTargetTool = defineTool({
1311
1423
  name: "record_plan_verification_target",
1312
1424
  label: "record_plan_verification_target",
1313
- description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Call once per target; entry carries id, type, commandLabel, file, requirementIds, uiStates, and optional symbol (omit it — the trace gate verifies the file and command, not a symbol). IMPORTANT: call this tool exactly ONE time per assistant message — never batch multiple record_* calls together in one message; emit one call, wait for its result, then call the next.",
1314
- promptSnippet: "Commit one plan verification target entry (one tool call per message).",
1425
+ description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Call once per target; entry carries id, type, commandLabel, file, requirementIds, uiStates, and optional symbol (omit it — the trace gate verifies the file and command, not a symbol). IMPORTANT: batch up to 5 record_* calls per assistant message (4-5 entries per message minimizes API round trips and rate-limit risk); never batch more than 5 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"<VT-XXX>\", \"type\": \"unit\", \"commandLabel\": \"<frozen command label>\", \"file\": \"<test file>\", \"requirementIds\": [\"<AC-XXX>\"], \"uiStates\": []}}",
1426
+ promptSnippet: "Commit 1-5 plan verification target entries (up to 5 per message).",
1315
1427
  parameters: Type.Object({
1316
1428
  entry: verificationTargetSchema,
1317
1429
  }, { additionalProperties: false }),
@@ -1324,6 +1436,64 @@ export async function createFrontendPlanLedgerTools(input) {
1324
1436
  error: "record_plan_verification_target requires a non-empty entry object",
1325
1437
  });
1326
1438
  }
1439
+ // Belt-and-braces for providers that do not strictly enforce the
1440
+ // tool-schema enum: reject an invalid type here with the allowed
1441
+ // values so the model can re-record in-node instead of the whole
1442
+ // attempt dying at compile time on the strict zod enum.
1443
+ const verificationTargetType = rawEntry.type;
1444
+ if (typeof verificationTargetType !== "string" ||
1445
+ !["static", "unit", "component", "integration", "mock"].includes(verificationTargetType)) {
1446
+ return planToolReceipt({
1447
+ ok: false,
1448
+ kind: "plan-verification-target",
1449
+ error: `record_plan_verification_target entry.type must be one of static | unit | component | integration | mock (received ${JSON.stringify(verificationTargetType ?? null)}); for a runtime behavior check use type=mock or type=integration`,
1450
+ });
1451
+ }
1452
+ // Duplicate-id rejection: committed typed facts are immutable, so
1453
+ // re-recording the same VT id would deadlock the compile with a
1454
+ // duplicate-id error the model cannot fix in-node. Reject here so
1455
+ // the model submits the correction under a fresh id.
1456
+ const vtId = typeof rawEntry.id === "string" ? rawEntry.id : "";
1457
+ if (vtId &&
1458
+ readCommittedEvents(store, attemptId).some((event) => {
1459
+ const fact = event.fact;
1460
+ if (!fact || fact.kind !== "plan-verification-target")
1461
+ return false;
1462
+ const entryFact = fact.entry;
1463
+ return entryFact?.id === vtId;
1464
+ })) {
1465
+ return planToolReceipt({
1466
+ ok: false,
1467
+ kind: "plan-verification-target",
1468
+ error: `record_plan_verification_target duplicate: verification target ${vtId} is already recorded; submit the corrected target under a new id instead`,
1469
+ });
1470
+ }
1471
+ // WriteSet containment at the boundary: a committed VT fact whose
1472
+ // file is outside the task writeSet is immutable, and the finalize
1473
+ // pre-validation would then fail the whole attempt with no in-node
1474
+ // cure (r17 post-merge). Reject here with the allowed patterns.
1475
+ if (typeof rawEntry.file === "string" &&
1476
+ input.writeSetPatterns &&
1477
+ input.writeSetPatterns.length > 0 &&
1478
+ !input.writeSetPatterns.some((pattern) => pathMatchesPattern(rawEntry.file, pattern))) {
1479
+ return planToolReceipt({
1480
+ ok: false,
1481
+ kind: "plan-verification-target",
1482
+ error: `record_plan_verification_target file is outside the writeSet patterns [${input.writeSetPatterns.join(", ")}]: ${rawEntry.file}; verification targets must point inside the task writeSet`,
1483
+ });
1484
+ }
1485
+ // Symbol shape check at the boundary: a fabricated symbol committed
1486
+ // here is immutable (duplicate ids are rejected), and while the
1487
+ // compile drops it, catching it now lets the model fix the target
1488
+ // in one receipt-free step.
1489
+ if (typeof rawEntry.symbol === "string" &&
1490
+ isSuspiciousVerificationSymbol(rawEntry.symbol)) {
1491
+ return planToolReceipt({
1492
+ ok: false,
1493
+ kind: "plan-verification-target",
1494
+ error: `record_plan_verification_target symbol "${rawEntry.symbol}" looks fabricated; use a real exported/describe/it symbol from ${rawEntry.file ?? "the target file"} or omit the symbol entirely (the trace gate verifies file+command)`,
1495
+ });
1496
+ }
1327
1497
  // uiStates: [] means this verification target is intentionally not
1328
1498
  // bound to a named UI state. Keep that canonical representation even
1329
1499
  // when a model omits the optional tool-boundary field.
@@ -1331,6 +1501,48 @@ export async function createFrontendPlanLedgerTools(input) {
1331
1501
  ...rawEntry,
1332
1502
  uiStates: stringList(rawEntry.uiStates),
1333
1503
  };
1504
+ // Cross-reference integrity at the boundary: the compile gate
1505
+ // rejects verification targets referencing UI states or
1506
+ // requirements that were never declared. Validate against the
1507
+ // facts already committed in this attempt so the model fixes the
1508
+ // reference in-node instead of burning the attempt at compile time
1509
+ // (r7: one full attempt lost to a single unknown UI state name).
1510
+ const committedEvents = readCommittedEvents(store, attemptId);
1511
+ const declaredUiStateNames = new Set(committedEvents.flatMap((event) => {
1512
+ const fact = event.fact;
1513
+ if (!fact || fact.kind !== "state-flow")
1514
+ return [];
1515
+ return (Array.isArray(fact.uiStates) ? fact.uiStates : [])
1516
+ .map((state) => isRecordObject(state) && typeof state.name === "string"
1517
+ ? state.name
1518
+ : "")
1519
+ .filter(Boolean);
1520
+ }));
1521
+ const unknownUiStates = entry.uiStates.filter((name) => !declaredUiStateNames.has(name));
1522
+ if (unknownUiStates.length > 0) {
1523
+ return planToolReceipt({
1524
+ ok: false,
1525
+ kind: "plan-verification-target",
1526
+ error: `record_plan_verification_target references UI states that were never declared: ${unknownUiStates.join(", ")}; declare every referenced UI state with record_state_flow first, or pass uiStates: [] for intentionally unbound targets`,
1527
+ });
1528
+ }
1529
+ const declaredRequirementIds = new Set(committedEvents.flatMap((event) => {
1530
+ const fact = event.fact;
1531
+ if (!fact || fact.kind !== "plan-requirement")
1532
+ return [];
1533
+ const requirementEntry = fact.entry;
1534
+ return typeof requirementEntry?.id === "string"
1535
+ ? [requirementEntry.id]
1536
+ : [];
1537
+ }));
1538
+ const unknownRequirementIds = stringList(rawEntry.requirementIds).filter((id) => !declaredRequirementIds.has(id));
1539
+ if (unknownRequirementIds.length > 0) {
1540
+ return planToolReceipt({
1541
+ ok: false,
1542
+ kind: "plan-verification-target",
1543
+ error: `record_plan_verification_target references unknown requirement ids: ${unknownRequirementIds.join(", ")}; record every referenced requirement with record_plan_requirement first`,
1544
+ });
1545
+ }
1334
1546
  const result = await adoptPlanFact("plan-verification-target", `${attemptId}:record_plan_verification_target:${randomUUID()}`, { kind: "plan-verification-target", origin: "plan", entry });
1335
1547
  return planToolReceipt(result);
1336
1548
  },
@@ -1338,8 +1550,8 @@ export async function createFrontendPlanLedgerTools(input) {
1338
1550
  const recordPlanEvidenceGapTool = defineTool({
1339
1551
  name: "record_plan_evidence_gap",
1340
1552
  label: "record_plan_evidence_gap",
1341
- description: "Commit one plan evidence gap entry (origin=plan plan-evidence-gap fact). Call once per gap; entry carries requirementId, description, blocking. IMPORTANT: call this tool exactly ONE time per assistant message — never batch multiple record_* calls together in one message; emit one call, wait for its result, then call the next.",
1342
- promptSnippet: "Commit one plan evidence gap entry (one tool call per message).",
1553
+ description: "Commit one plan evidence gap entry (origin=plan plan-evidence-gap fact). Call once per gap; entry carries requirementId, description, blocking. IMPORTANT: batch up to 5 record_* calls per assistant message (4-5 entries per message minimizes API round trips and rate-limit risk); never batch more than 5 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"requirementId\": \"<AC-XXX>\", \"description\": \"<what evidence is missing and why>\", \"blocking\": false}}",
1554
+ promptSnippet: "Commit 1-5 plan evidence gap entries (up to 5 per message).",
1343
1555
  parameters: Type.Object({
1344
1556
  entry: evidenceGapSchema,
1345
1557
  }, { additionalProperties: false }),
@@ -1353,6 +1565,18 @@ export async function createFrontendPlanLedgerTools(input) {
1353
1565
  });
1354
1566
  }
1355
1567
  const entry = rawEntry;
1568
+ // A standalone evidence gap IS the gap statement: a blank description
1569
+ // would fail the canonical contract schema (min 1 char) after the
1570
+ // whole attempt finished. Reject at the boundary so the model writes
1571
+ // a real description in-node instead of burning the attempt.
1572
+ if (typeof entry.description === "string" &&
1573
+ entry.description.trim() === "") {
1574
+ return planToolReceipt({
1575
+ ok: false,
1576
+ kind: "plan-evidence-gap",
1577
+ error: "record_plan_evidence_gap requires a non-empty description describing the gap",
1578
+ });
1579
+ }
1356
1580
  const result = await adoptPlanFact("plan-evidence-gap", `${attemptId}:record_plan_evidence_gap:${randomUUID()}`, { kind: "plan-evidence-gap", origin: "plan", entry });
1357
1581
  return planToolReceipt(result);
1358
1582
  },
@@ -1385,6 +1609,66 @@ export async function createFrontendPlanLedgerTools(input) {
1385
1609
  ? { realIntegrationGap: params.realIntegrationGap }
1386
1610
  : {}),
1387
1611
  };
1612
+ // Front-load the node's compile + policy gates into the finalize
1613
+ // receipt (same pipeline the design-policy shell and the node
1614
+ // self-check run: merge the patch onto the runtime skeleton,
1615
+ // then the full analyze). A failing gate used to burn an entire
1616
+ // attempt per finding (r8/r9: ui-design-coverage, verification
1617
+ // targets, UI-state shape, one attempt each); surfaced here the
1618
+ // model fixes the facts and re-calls finalize_plan in-node.
1619
+ if (input.skeleton && input.sourceBinding) {
1620
+ // Front-load the exact pipeline the design-policy shell and
1621
+ // the node self-check run (patch ⊕ skeleton -> analyze ->
1622
+ // policy pre-checks) into the finalize receipt. Findings
1623
+ // come back as fixable receipt errors instead of burning
1624
+ // an attempt per gate (r8/r9: coverage, verification
1625
+ // targets, UI-state shape each cost a full attempt).
1626
+ const { analyzeFrontendPlanPatchCandidate, applyFrontendContractMergePatch, FrontendContractFailure, PlanPolicyPrecheckFailure, serializeDeterministicJson, } = await import("../workflows/dag/frontend-implementation-contract.js");
1627
+ try {
1628
+ const merged = applyFrontendContractMergePatch(input.skeleton, patch);
1629
+ await analyzeFrontendPlanPatchCandidate({
1630
+ runDir: input.runDir,
1631
+ rawContractText: serializeDeterministicJson(merged),
1632
+ sourceBinding: input.sourceBinding,
1633
+ });
1634
+ }
1635
+ catch (error) {
1636
+ if (error instanceof PlanPolicyPrecheckFailure) {
1637
+ // Template the fix: every uncovered interaction / state
1638
+ // maps to a ready-to-submit record_component_choice
1639
+ // call. One reuse-existing choice covers all
1640
+ // behavioural interactions.
1641
+ const suggestions = error.findings
1642
+ .filter((finding) => finding.code === "ui-design-coverage-missing" &&
1643
+ finding.path)
1644
+ .map((finding) => ({
1645
+ tool: "record_component_choice",
1646
+ args: {
1647
+ choice: {
1648
+ purpose: finding.path,
1649
+ component: "<name the existing or new component>",
1650
+ decision: "reuse-existing",
1651
+ },
1652
+ },
1653
+ }));
1654
+ const suggestionBlock = suggestions.length > 0
1655
+ ? ` Suggested record_* calls (copy, fill component, submit): ${JSON.stringify(suggestions)}`
1656
+ : "";
1657
+ return planToolReceipt({
1658
+ ok: false,
1659
+ kind: "finalize_plan",
1660
+ error: `finalize_plan pre-validation failed (fix the listed plan facts with record_* tools, then call finalize_plan again): ${error.message}${suggestionBlock}`,
1661
+ });
1662
+ }
1663
+ if (!(error instanceof FrontendContractFailure))
1664
+ throw error;
1665
+ return planToolReceipt({
1666
+ ok: false,
1667
+ kind: "finalize_plan",
1668
+ error: `finalize_plan pre-validation failed (fix the listed plan facts with record_* tools, then call finalize_plan again): ${error.message}`,
1669
+ });
1670
+ }
1671
+ }
1388
1672
  const patchResult = await adoptPlanFact("target-surface", `${attemptId}:finalize_plan:patch:${randomUUID()}`, { kind: "target-surface", origin: "plan", patch });
1389
1673
  if (!patchResult.ok) {
1390
1674
  return planToolReceipt({
@@ -1464,6 +1748,19 @@ export async function createFrontendPlanLedgerTools(input) {
1464
1748
  const committed = readCommittedEvents(store, attemptId);
1465
1749
  await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"), committed);
1466
1750
  },
1751
+ committedFactCount: () => readCommittedEvents(store, attemptId).length,
1752
+ committedRequirementIds: () => {
1753
+ const ids = new Set();
1754
+ for (const event of readCommittedEvents(store, attemptId)) {
1755
+ const fact = event.fact;
1756
+ if (!fact || fact.kind !== "plan-requirement")
1757
+ continue;
1758
+ const entryFact = fact.entry;
1759
+ if (typeof entryFact?.id === "string")
1760
+ ids.add(entryFact.id);
1761
+ }
1762
+ return ids;
1763
+ },
1467
1764
  };
1468
1765
  }
1469
1766
  function isRecordObject(value) {
@@ -1597,8 +1894,8 @@ export async function createFrontendContractTools(input) {
1597
1894
  const recordTools = Object.entries(recordKinds).map(([name, kind]) => defineTool({
1598
1895
  name,
1599
1896
  label: name,
1600
- description: `Commit an origin=contract ${kind} fact.`,
1601
- promptSnippet: `Commit an origin=contract ${kind} fact.`,
1897
+ description: `Commit an origin=contract ${kind} fact. IMPORTANT: submit incrementally — batch up to 5 record_* calls per message, starting from the FIRST message; never attempt to emit the whole contract in one response (a single large dump will be truncated and rejected). Every message must make progress by committing at least one record_* fact.`,
1898
+ promptSnippet: `Commit 1-5 origin=contract ${kind} facts (up to 5 per message).`,
1602
1899
  parameters: Type.Object({}, { additionalProperties: true }),
1603
1900
  async execute(_toolCallId, params) {
1604
1901
  const result = await adoptContractFact(kind, {
@@ -1753,6 +2050,22 @@ export async function createFrontendScoutEvidenceTools(input) {
1753
2050
  continue;
1754
2051
  }
1755
2052
  try {
2053
+ // Directories are legitimate named targets (greenfield smoke: the
2054
+ // page directory exists while the files inside it are to be
2055
+ // created). readFile on a directory throws EISDIR, which used to
2056
+ // mark every directory path fresh=false and structurally fail the
2057
+ // freshness gate for create-new surfaces. stat() first: a directory
2058
+ // counts as fresh existence evidence; its content hash is a stable
2059
+ // directory marker since there is no single file content to hash.
2060
+ const info = await stat(absolute);
2061
+ if (info.isDirectory()) {
2062
+ evidence.push({
2063
+ path: relative,
2064
+ sha256: createHash("sha256").update(`directory:${relative}`).digest("hex"),
2065
+ fresh: true,
2066
+ });
2067
+ continue;
2068
+ }
1756
2069
  const bytes = await readFile(absolute);
1757
2070
  evidence.push({
1758
2071
  path: relative,
@@ -1803,7 +2116,7 @@ export async function createFrontendScoutEvidenceTools(input) {
1803
2116
  const recordTargetSurfaceTool = defineTool({
1804
2117
  name: "record_target_surface",
1805
2118
  label: "record_target_surface",
1806
- description: "Commit an origin=scout target-surface fact with complete/blocked discovery status. A complete surface needs a proven target path and no unresolved paths; blocked surfaces name the unresolved paths instead of guessing.",
2119
+ description: "Commit an origin=scout target-surface fact with complete/blocked discovery status. A complete surface needs a proven target path and no unresolved paths; blocked surfaces name the unresolved paths instead of guessing. Example: {\"completeness\": \"complete\", \"entrypoint\": \"<file>\", \"implementationPaths\": [\"<dir or file>\"], \"testPaths\": [\"<file>\"], \"allowedPathConflicts\": [], \"unresolvedPaths\": []}",
1807
2120
  promptSnippet: "Commit an origin=scout target-surface fact.",
1808
2121
  parameters: Type.Object({
1809
2122
  completeness: scoutCompleteness,
@@ -1842,7 +2155,7 @@ export async function createFrontendScoutEvidenceTools(input) {
1842
2155
  const recordDesignEvidenceTool = defineTool({
1843
2156
  name: "record_design_evidence",
1844
2157
  label: "record_design_evidence",
1845
- description: "Commit an origin=scout design-evidence fact (source, paths, conflicts).",
2158
+ description: "Commit an origin=scout design-evidence fact (source, paths, conflicts). Example: {\"source\": \"<source>\", \"paths\": [\"<file>\"], \"conflicts\": []}",
1846
2159
  promptSnippet: "Commit an origin=scout design-evidence fact.",
1847
2160
  parameters: Type.Object({ source: Type.String({}), paths: stringArray, conflicts: stringArray }, { additionalProperties: false }),
1848
2161
  async execute(_toolCallId, params) {
@@ -2239,6 +2552,171 @@ async function runFrontendDesignTerminalShadow(input) {
2239
2552
  }
2240
2553
  return input.mapped;
2241
2554
  }
2555
+ /** Tool subsets for the three frontend plan sessions (r17 split design). */
2556
+ const FRONTEND_PLAN_SEGMENTS = [
2557
+ {
2558
+ id: "coverage",
2559
+ toolNames: new Set([
2560
+ "record_plan_requirement",
2561
+ "record_plan_verification_target",
2562
+ "adopt_staged_fact",
2563
+ ]),
2564
+ instruction: [
2565
+ "PLAN SEGMENT 1/3 — coverage mapping only.",
2566
+ "Your ONLY job: for every frozen requirement, emit record_plan_requirement (requirement → implementation files) and record_plan_verification_target (verification target bound to requirement ids and files). Do NOT record components, UI states, mock, dependency, or routes — a follow-up session owns those.",
2567
+ "Do not call finalize_plan; it is not available in this segment.",
2568
+ ].join(" "),
2569
+ },
2570
+ {
2571
+ id: "ux-decisions",
2572
+ toolNames: new Set([
2573
+ "record_component_choice",
2574
+ "record_state_flow",
2575
+ "record_data_flow",
2576
+ "record_mock_api",
2577
+ "record_design_deviation",
2578
+ "record_dependency",
2579
+ "record_route_selection",
2580
+ "adopt_staged_fact",
2581
+ ]),
2582
+ instruction: [
2583
+ "PLAN SEGMENT 2/3 — UX decisions.",
2584
+ "Requirements and verification targets are already committed in the ledger (do NOT re-record them; duplicates are rejected). Your ONLY job: record component choices (decision=new requires sourceRequirementIds per the citation table), state flows, data flow, mock strategy, design deviation, dependency policy, and route selection.",
2585
+ "Do not call finalize_plan; it is not available in this segment.",
2586
+ ].join(" "),
2587
+ },
2588
+ {
2589
+ id: "finalize",
2590
+ toolNames: null,
2591
+ instruction: [
2592
+ "PLAN SEGMENT 3/3 — finalize.",
2593
+ "All record_* tools are available: if a finalize_plan receipt reports missing or invalid facts, fix them with the named record_* tool and call finalize_plan again. Otherwise call finalize_plan exactly once with no extra fields.",
2594
+ ].join(" "),
2595
+ },
2596
+ ];
2597
+ /**
2598
+ * Coverage-batch slicing for the frontend plan split (options 1+5): the
2599
+ * coverage segment becomes one session per requirement slice (default 4
2600
+ * requirements), so a small output budget can never be exhausted by
2601
+ * upfront reasoning about the whole requirement list. Zero-progress
2602
+ * batches are split in half and retried (option 5); single-requirement
2603
+ * zero-progress failures short-circuit to the retry ladder.
2604
+ */
2605
+ const FRONTEND_PLAN_COVERAGE_BATCH_SIZE = 4;
2606
+ const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 32;
2607
+ export async function runFrontendPlanSegmentedSessions(input) {
2608
+ const queue = [];
2609
+ // An empty list means "ledger unreadable / unknown" and must fall back to
2610
+ // one unscoped coverage session — only a non-empty list batches.
2611
+ const requirementIdsProvided = input.requirementIds !== undefined && input.requirementIds.length > 0;
2612
+ const pending = (input.requirementIds ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
2613
+ for (const segment of FRONTEND_PLAN_SEGMENTS) {
2614
+ if (segment.id !== "coverage") {
2615
+ queue.push({
2616
+ id: segment.id,
2617
+ toolNames: segment.toolNames,
2618
+ prompt: `${input.basePrompt}\n\n${segment.instruction}`,
2619
+ });
2620
+ continue;
2621
+ }
2622
+ // No requirement list (unreadable ledger) -> one unscoped coverage
2623
+ // session. A provided list with everything committed (resume) skips
2624
+ // coverage entirely.
2625
+ if (!requirementIdsProvided || pending.length > 0) {
2626
+ if (!requirementIdsProvided) {
2627
+ queue.push({
2628
+ id: segment.id,
2629
+ toolNames: segment.toolNames,
2630
+ prompt: `${input.basePrompt}\n\n${segment.instruction}`,
2631
+ });
2632
+ continue;
2633
+ }
2634
+ for (let i = 0; i < pending.length; i += FRONTEND_PLAN_COVERAGE_BATCH_SIZE) {
2635
+ const slice = pending.slice(i, i + FRONTEND_PLAN_COVERAGE_BATCH_SIZE);
2636
+ queue.push({
2637
+ id: `coverage-batch-${i / FRONTEND_PLAN_COVERAGE_BATCH_SIZE + 1}`,
2638
+ toolNames: segment.toolNames,
2639
+ coverageSlice: slice,
2640
+ prompt: `${input.basePrompt}\n\n${segment.instruction}\n\nCOVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them.`,
2641
+ });
2642
+ }
2643
+ }
2644
+ }
2645
+ let last;
2646
+ let index = 0;
2647
+ while (index < queue.length && index < FRONTEND_PLAN_BATCH_MAX_SESSIONS) {
2648
+ const session = queue[index];
2649
+ const remaining = (session.coverageSlice ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
2650
+ // Resume/earlier-batch commits may already cover this slice.
2651
+ if (session.coverageSlice && remaining.length === 0) {
2652
+ index += 1;
2653
+ continue;
2654
+ }
2655
+ let prompt = session.prompt;
2656
+ if (session.coverageSlice) {
2657
+ prompt = prompt.replace(/COVERAGE BATCH: process ONLY these requirements in this session: .*/, `COVERAGE BATCH: process ONLY these requirements in this session: ${remaining.join(", ")}. Other requirements are handled by separate sessions; do not record them.`);
2658
+ }
2659
+ const committedBefore = input.committedFactCount();
2660
+ const customTools = input.segmentCustomTools(session.toolNames);
2661
+ const result = await input.piStepFn({
2662
+ ...input.sessionOptions,
2663
+ prompt,
2664
+ ...(customTools.length > 0
2665
+ ? {
2666
+ writerToolPolicy: {
2667
+ requireSdk: true,
2668
+ customTools,
2669
+ },
2670
+ }
2671
+ : {}),
2672
+ });
2673
+ last = result;
2674
+ try {
2675
+ await input.flushLedger();
2676
+ }
2677
+ catch {
2678
+ // best-effort: the node-level flush runs again after the attempt
2679
+ }
2680
+ const committedAfter = input.committedFactCount();
2681
+ if (result.ok) {
2682
+ index += 1;
2683
+ continue;
2684
+ }
2685
+ const committedFactsOnlySuccess = session.id !== "finalize" &&
2686
+ !(result.assistantText ?? "").trim() &&
2687
+ !result.stderr.trim() &&
2688
+ !result.timedOut &&
2689
+ committedAfter > committedBefore;
2690
+ if (committedFactsOnlySuccess) {
2691
+ // The session died but banked facts: keep the progress and move on.
2692
+ index += 1;
2693
+ continue;
2694
+ }
2695
+ // Option 5: a multi-requirement coverage batch that failed with ZERO
2696
+ // new facts and no provider stderr is the upfront-reasoning burn —
2697
+ // halve the slice and retry instead of failing the attempt.
2698
+ const coverageSlice = session.coverageSlice;
2699
+ const zeroProgressBurn = coverageSlice !== undefined &&
2700
+ coverageSlice.length > 1 &&
2701
+ committedAfter === committedBefore &&
2702
+ !(result.assistantText ?? "").trim() &&
2703
+ !result.stderr.trim() &&
2704
+ !result.timedOut;
2705
+ if (zeroProgressBurn && coverageSlice) {
2706
+ const half = Math.ceil(coverageSlice.length / 2);
2707
+ queue.splice(index, 1, { ...session, coverageSlice: coverageSlice.slice(0, half) }, { ...session, coverageSlice: coverageSlice.slice(half) });
2708
+ continue;
2709
+ }
2710
+ return result;
2711
+ }
2712
+ return (last ?? {
2713
+ ok: false,
2714
+ stdout: "",
2715
+ stderr: "frontend plan segmentation produced no session",
2716
+ failureCategory: "empty-output",
2717
+ durationMs: 0,
2718
+ });
2719
+ }
2242
2720
  export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
2243
2721
  const started = Date.now();
2244
2722
  const persona = resolveDagPiPersona(input.task);
@@ -2524,6 +3002,8 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
2524
3002
  runDir: meta.runDir,
2525
3003
  nodeId: input.task.id,
2526
3004
  skeleton: input.task.structuredContractOutput?.skeleton,
3005
+ sourceBinding: meta.spec.sourceBinding,
3006
+ writeSetPatterns: input.task.writeSet,
2527
3007
  componentNewSourceReferences: await resolveFrontendPlanNewComponentSourceReferences({
2528
3008
  cwd: input.cwd,
2529
3009
  sourceBinding: meta.spec.sourceBinding,
@@ -2619,10 +3099,9 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
2619
3099
  const piExtensionPaths = piExtensionsResolution && resolvePiBackend() !== "cli-only"
2620
3100
  ? piExtensionsResolution.resolved.flatMap((entry) => entry.entryPaths)
2621
3101
  : undefined;
2622
- result = await piStepFn({
3102
+ const piSessionOptions = {
2623
3103
  attachedFiles: [],
2624
3104
  modelConfig: resolveDagPiModelConfig(input.model, input.thinking ? { thinking: input.thinking } : undefined),
2625
- prompt: input.prompt,
2626
3105
  repoRoot: input.cwd,
2627
3106
  step,
2628
3107
  toolNames: resolveDagPiToolNames(input.task),
@@ -2635,7 +3114,6 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
2635
3114
  ...(input.task.contextBudget
2636
3115
  ? { contextBudget: input.task.contextBudget }
2637
3116
  : {}),
2638
- ...(writerToolPolicy ? { writerToolPolicy } : {}),
2639
3117
  ...(piExtensionPaths && piExtensionPaths.length > 0
2640
3118
  ? { piExtensionPaths }
2641
3119
  : {}),
@@ -2648,7 +3126,55 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
2648
3126
  bridgeActivity(activity.kind, activity.at);
2649
3127
  }
2650
3128
  : undefined,
2651
- });
3129
+ };
3130
+ if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
3131
+ // Frontend-only split: three sequential sessions with independent
3132
+ // output budgets (coverage -> UX decisions -> finalize), mirroring the
3133
+ // backend-test template's module sharding. Every other template keeps
3134
+ // the single-session path below.
3135
+ let planRequirementIds = [];
3136
+ try {
3137
+ const { readCommittedOriginFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
3138
+ const contractFacts = await readCommittedOriginFacts(meta.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl");
3139
+ // Only requirement facts: the contract ledger also carries
3140
+ // constraints (CON-*), evidence expectations (EV-*), handoff
3141
+ // intents (HND-*), open questions (OQ-*) and split proposals
3142
+ // (SPLIT-*) that all have ids — feeding those into the coverage
3143
+ // batches made the model record non-frozen plan-requirement ids
3144
+ // that finalize's canonical-coverage gate then rejected (r-ext2).
3145
+ planRequirementIds = contractFacts
3146
+ .filter((record) => record.fact
3147
+ ?.kind === "requirement")
3148
+ .map((record) => record.fact?.id)
3149
+ .filter((id) => typeof id === "string")
3150
+ .sort();
3151
+ }
3152
+ catch {
3153
+ // Unreadable ledger falls back to a single coverage session.
3154
+ }
3155
+ result = await runFrontendPlanSegmentedSessions({
3156
+ piStepFn,
3157
+ sessionOptions: piSessionOptions,
3158
+ basePrompt: input.prompt,
3159
+ attempt: input.attempt ?? 1,
3160
+ committedFactCount: () => planLedgerTools.committedFactCount(),
3161
+ requirementIds: planRequirementIds,
3162
+ committedRequirementIds: () => planLedgerTools.committedRequirementIds(),
3163
+ segmentCustomTools: (toolNames) => toolNames === null
3164
+ ? planLedgerTools.customTools
3165
+ : planLedgerTools.customTools.filter((tool) => typeof tool === "object" &&
3166
+ tool !== null &&
3167
+ toolNames.has(tool.name)),
3168
+ flushLedger: () => planLedgerTools.flush(),
3169
+ });
3170
+ }
3171
+ else {
3172
+ result = await piStepFn({
3173
+ ...piSessionOptions,
3174
+ prompt: input.prompt,
3175
+ ...(writerToolPolicy ? { writerToolPolicy } : {}),
3176
+ });
3177
+ }
2652
3178
  }
2653
3179
  catch (error) {
2654
3180
  if (playwrightToolContext) {
@@ -2785,27 +3311,28 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
2785
3311
  catch {
2786
3312
  // best-effort flush; missing ledger still fails at the node validator
2787
3313
  }
2788
- // Read-burst guard: a plan that burned dozens of read/grep/ls/find
2789
- // calls (re-reading upstream outputs and source files it should trust
2790
- // from typed facts) blows up the context window and eventually fails
2791
- // with 400 request-too-large. Detect it deterministically from the
2792
- // session log and retry with a reduced-reading instruction.
2793
- const readBurstIssues = input.task.readBudget?.onExhaustion === "return-guidance"
2794
- ? []
2795
- : await detectPlanReadBurst({
2796
- runDir: meta.runDir,
2797
- nodeId: input.task.id,
2798
- });
2799
- if (readBurstIssues.length > 0) {
2800
- return {
2801
- ...mapped,
2802
- ok: false,
2803
- failureCategory: "read-burst",
2804
- stderr: [mapped.stderr, ...readBurstIssues].filter(Boolean).join("\n\n"),
2805
- };
2806
- }
3314
+ // NOTE: the legacy plan read-burst guard lived here. It is dead code
3315
+ // since the plan node went tool-only (resolveDagPiToolNames grants no
3316
+ // read/grep/ls/find), so a read burst is structurally impossible; the
3317
+ // read-budget path (scout etc.) keeps its own guard.
2807
3318
  }
2808
3319
  if (!isWriteTask) {
3320
+ if (!mapped.ok &&
3321
+ !(mapped.assistantText ?? "").trim() &&
3322
+ !mapped.stderr.trim()) {
3323
+ // Terminal-fact acceptance: the run finished with every fact committed
3324
+ // (including the terminal) but no final narrative text. A timeout or
3325
+ // provider error always leaves supervision/provider stderr, so a
3326
+ // blank stderr here means the only "failure" is the empty text.
3327
+ const terminalAccepted = await acceptCommittedTypedTerminalFact(meta.runDir, input.task.id);
3328
+ if (terminalAccepted) {
3329
+ return {
3330
+ ...mapped,
3331
+ ok: true,
3332
+ failureCategory: undefined,
3333
+ };
3334
+ }
3335
+ }
2809
3336
  return mapped;
2810
3337
  }
2811
3338
  let writeGuardOk = true;