@tea-agent/loop-agent 0.39.0-beta.13 → 0.39.0-beta.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -16,6 +16,7 @@ import { redactPromptForLog, truncateOutput, } from "../shared/output-truncation
16
16
  import { GitStatusUnavailableError, pathsChangedDuringRun, readGitStatusPorcelain, recoverRootNulArtifact, snapshotGitStatusPathFingerprints, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
17
17
  import { captureWorkspaceWriteSnapshot, diffWorkspaceWriteSnapshots, } from "./workspace-write-snapshot.js";
18
18
  import { pathMatchesPattern } from "../shared/git-progress.js";
19
+ import { isSuspiciousVerificationSymbol } from "../workflows/dag/frontend-implementation-contract.js";
19
20
  import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
20
21
  import { assessBackendTestMdPlanCompleteness, assessBackendTestMdWriterCompleteness, assessBackendTestPytestPlanCompleteness, assessBackendTestPytestWriterCompleteness, assessBackendTestShardChildCompleteness, backendTestWriterProgressRoleForTask, classifyBackendTestWriterCompletenessFailure, isBackendTestCompletenessRetryCandidate, isBackendTestMdPlanTask, isBackendTestPytestPlanTask, isBackendTestShardChildTask, writeBackendTestWriterProgressArtifacts, } from "../workflows/dag/backend-test-writer-completeness.js";
21
22
  import { resolveBackendTestLayout } from "../workflows/dag/backend-test-layout.js";
@@ -236,16 +237,6 @@ export const FRONTEND_PLAN_RECORD_TOOL_NAMES = [
236
237
  ];
237
238
  export const FRONTEND_PLAN_TERMINAL_TOOL_NAMES = ["finalize_plan"];
238
239
  export const FRONTEND_PLAN_ADOPT_TOOL_NAMES = ["adopt_staged_fact"];
239
- /**
240
- * Inter-call delay for high-frequency frontend record_* tools. Incremental
241
- * submission (one model response per 1-5 entries) means dozens of sequential
242
- * API round trips in a few minutes; third-party gateways (huu.dqy.ink) trip
243
- * rate limits whose windows exceed normal API RPM even at ~13 calls/min. A
244
- * short fixed delay before each tool body keeps the sustained rate under
245
- * typical thresholds while batching cuts the total call count.
246
- */
247
- const FRONTEND_RECORD_TOOL_THROTTLE_MS = 1000;
248
- const sleepRecordThrottle = () => new Promise((resolve) => setTimeout(resolve, FRONTEND_RECORD_TOOL_THROTTLE_MS));
249
240
  /** M5: `frontend-review-pi` emits its authoritative terminal verdict through
250
241
  * committed typed tools instead of the legacy JSON verdict parse. */
251
242
  export function isFrontendReviewTypedTerminalNode(task) {
@@ -395,10 +386,6 @@ export function scanReviewTerminalKindsFromSessionEvents(content) {
395
386
  }
396
387
  return kinds;
397
388
  }
398
- /** Read-only discovery tool budget for frontend-plan-pi. Exceeding it means
399
- * the plan re-read upstream outputs/sources instead of trusting typed facts,
400
- * which blows up the context window (400 request-too-large). */
401
- const PLAN_READ_TOOL_BUDGET = 40;
402
389
  const READ_ONLY_TOOL_NAMES = new Set(["read", "grep", "ls", "find"]);
403
390
  function readEventPath(event) {
404
391
  const candidates = [event.path, event.readPath, event.input];
@@ -483,45 +470,6 @@ export async function detectNodeReadBudget(input) {
483
470
  issues.push(`frontend ${input.nodeId} read budget exceeded: ${stats.elapsedMs}ms (budget ${input.budget.maxMs}ms)`);
484
471
  return issues;
485
472
  }
486
- /** Deterministic read-burst guard for frontend-plan-pi: count read-only
487
- * discovery tool calls (read/grep/ls/find) from the session log. Over budget →
488
- * read-burst, retried with a reduced-reading instruction. Pure scan; a
489
- * successful plan under budget is never blocked. */
490
- export async function detectPlanReadBurst(input) {
491
- const sessionEventsPath = path.join(input.runDir, input.nodeId, "session-events.jsonl");
492
- let count = 0;
493
- try {
494
- const content = await readFile(sessionEventsPath, "utf8");
495
- for (const line of content.split("\n")) {
496
- if (!line.trim())
497
- continue;
498
- try {
499
- const event = JSON.parse(line);
500
- if (event.type === "tool_execution_start" &&
501
- typeof event.toolName === "string" &&
502
- (event.toolName === "read" ||
503
- event.toolName === "grep" ||
504
- event.toolName === "ls" ||
505
- event.toolName === "find")) {
506
- count += 1;
507
- }
508
- }
509
- catch {
510
- // skip unparseable line
511
- }
512
- }
513
- }
514
- catch {
515
- // Missing/unreadable session log → no burst detection
516
- return [];
517
- }
518
- if (count > PLAN_READ_TOOL_BUDGET) {
519
- return [
520
- `frontend plan read-burst: ${count} read-only tool calls (budget ${PLAN_READ_TOOL_BUDGET}). Trust the upstream contract/scout typed facts; do not re-read contract/scout outputs or source files already captured. Minimize discovery reads, commit record_* facts directly, then finalize_plan.`,
521
- ];
522
- }
523
- return [];
524
- }
525
473
  export const FRONTEND_DESIGN_TERMINAL_TOOL_NAMES = new Set([
526
474
  "approve_design",
527
475
  "request_design_changes",
@@ -932,11 +880,27 @@ export async function createFrontendPlanLedgerTools(input) {
932
880
  import("typebox"),
933
881
  import("@earendil-works/pi-coding-agent"),
934
882
  ]);
935
- const { readCommittedEvents, writeTypedEventStoreJsonl } = await import("../workflows/dag/frontend-typed-event-store.js");
883
+ const { loadTypedEventStore, readCommittedEvents, writeTypedEventStoreJsonl, } = await import("../workflows/dag/frontend-typed-event-store.js");
936
884
  const { adoptStagedFact, adoptTypedEventFact, stageTypedEventFact } = await import("../workflows/dag/frontend-typed-event-transaction.js");
937
885
  const { assemblePlanPatchFromCommittedFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
938
886
  const store = input.store;
939
887
  const attemptId = input.attemptId;
888
+ // A retry creates a fresh executor-local store, but the plan ledger is the
889
+ // cross-attempt authority. Restore the committed prefix before registering
890
+ // tools; otherwise the first flush of a retry can overwrite facts that the
891
+ // previous attempt had already committed. The on-disk file contains only
892
+ // committed records, so loading it is also fail-closed with respect to
893
+ // staged/quarantined facts.
894
+ const persisted = await loadTypedEventStore(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"));
895
+ if (persisted.records.length > 0) {
896
+ const existingEventIds = new Set(store.records.map((record) => record.eventId));
897
+ for (const record of persisted.records) {
898
+ if (!existingEventIds.has(record.eventId)) {
899
+ store.records.push(record);
900
+ }
901
+ }
902
+ store.revision = Math.max(store.revision, persisted.revision);
903
+ }
940
904
  const stringArray = Type.Array(Type.String({}));
941
905
  const optionalString = Type.Optional(Type.String({}));
942
906
  const optionalStringArray = Type.Optional(stringArray);
@@ -1127,7 +1091,6 @@ export async function createFrontendPlanLedgerTools(input) {
1127
1091
  stylingStrategy: optionalString,
1128
1092
  }, { additionalProperties: false }),
1129
1093
  async execute(_toolCallId, params) {
1130
- await sleepRecordThrottle();
1131
1094
  const rawChoice = params?.choice;
1132
1095
  if (!isRecordObject(rawChoice)) {
1133
1096
  return planToolReceipt({
@@ -1180,7 +1143,16 @@ export async function createFrontendPlanLedgerTools(input) {
1180
1143
  ? { stylingStrategy: params.stylingStrategy }
1181
1144
  : {}),
1182
1145
  });
1183
- return planToolReceipt(result);
1146
+ // Echo the frozen citation the runtime derived: the model sees the
1147
+ // purpose↔citation mapping it just committed and can re-record the
1148
+ // choice (last-wins per purpose at compile) when it mismatches.
1149
+ const echo = {
1150
+ ...result,
1151
+ ...(choice.specReference
1152
+ ? { derivedSpecReference: choice.specReference }
1153
+ : {}),
1154
+ };
1155
+ return planToolReceipt(echo);
1184
1156
  },
1185
1157
  });
1186
1158
  const recordStateFlowTool = defineTool({
@@ -1264,6 +1236,26 @@ export async function createFrontendPlanLedgerTools(input) {
1264
1236
  error: "record_state_flow requires non-empty interaction.name (id is accepted only as a legacy alias)",
1265
1237
  });
1266
1238
  }
1239
+ // Interaction -> VT forward references are legal only against
1240
+ // already-committed VT facts. In the segmented flow every VT
1241
+ // commits in the coverage segment before state flows run, so a
1242
+ // dangling reference here is a real defect (r20: *-BEHAVIOR
1243
+ // refs reached the final review untraceable).
1244
+ const interactionVtIds = stringList(interaction.verificationTargetIds);
1245
+ const committedVtIds = new Set(readCommittedEvents(store, attemptId)
1246
+ .map((event) => event.fact)
1247
+ .filter((fact) => fact.kind === "plan-verification-target")
1248
+ .map((fact) => fact.entry
1249
+ ?.id)
1250
+ .filter((id) => typeof id === "string"));
1251
+ const unknownVtIds = interactionVtIds.filter((id) => !committedVtIds.has(id));
1252
+ if (unknownVtIds.length > 0) {
1253
+ return planToolReceipt({
1254
+ ok: false,
1255
+ kind: "state-flow",
1256
+ error: `record_state_flow interaction "${resolvedName}" references verification targets that are not recorded yet: ${unknownVtIds.join(", ")}; record them with record_plan_verification_target first, then re-record this state flow`,
1257
+ });
1258
+ }
1267
1259
  interactions.push({ ...interaction, name: resolvedName });
1268
1260
  }
1269
1261
  const states = uiStates
@@ -1359,7 +1351,6 @@ export async function createFrontendPlanLedgerTools(input) {
1359
1351
  entry: requirementSchema,
1360
1352
  }, { additionalProperties: false }),
1361
1353
  async execute(_toolCallId, params) {
1362
- await sleepRecordThrottle();
1363
1354
  const rawEntry = params?.entry;
1364
1355
  if (!isRecordObject(rawEntry)) {
1365
1356
  return planToolReceipt({
@@ -1387,11 +1378,26 @@ export async function createFrontendPlanLedgerTools(input) {
1387
1378
  void _omittedGap;
1388
1379
  entry = rest;
1389
1380
  }
1381
+ // Canonical-identity check: a requirement id outside the frozen
1382
+ // canonical list (e.g. a BR-* business rule picked up from the PRD
1383
+ // prose) would commit an immutable fact that finalize's
1384
+ // canonical-coverage gate rejects with no in-node cure. Reject here
1385
+ // and name the allowed ids.
1386
+ const id = typeof entry.id === "string" ? entry.id : "";
1387
+ if (id &&
1388
+ input.requirementIds &&
1389
+ input.requirementIds.length > 0 &&
1390
+ !input.requirementIds.includes(id)) {
1391
+ return planToolReceipt({
1392
+ ok: false,
1393
+ kind: "plan-requirement",
1394
+ error: `record_plan_requirement id "${id}" is not a frozen canonical requirement; canonical ids are: ${input.requirementIds.join(", ")}`,
1395
+ });
1396
+ }
1390
1397
  // A requirement id is a canonical identity: recording it twice would
1391
1398
  // compile a duplicate requirements[] entry and fail design review.
1392
1399
  // Reject duplicates at the tool boundary so the model can fix them
1393
1400
  // in-node instead of burning the attempt on a later validation error.
1394
- const id = typeof entry.id === "string" ? entry.id : "";
1395
1401
  if (id) {
1396
1402
  const existing = readCommittedEvents(store, attemptId).find((event) => {
1397
1403
  const fact = event.fact;
@@ -1422,7 +1428,6 @@ export async function createFrontendPlanLedgerTools(input) {
1422
1428
  entry: verificationTargetSchema,
1423
1429
  }, { additionalProperties: false }),
1424
1430
  async execute(_toolCallId, params) {
1425
- await sleepRecordThrottle();
1426
1431
  const rawEntry = params?.entry;
1427
1432
  if (!isRecordObject(rawEntry)) {
1428
1433
  return planToolReceipt({
@@ -1477,6 +1482,18 @@ export async function createFrontendPlanLedgerTools(input) {
1477
1482
  error: `record_plan_verification_target file is outside the writeSet patterns [${input.writeSetPatterns.join(", ")}]: ${rawEntry.file}; verification targets must point inside the task writeSet`,
1478
1483
  });
1479
1484
  }
1485
+ // Symbol shape check at the boundary: a fabricated symbol committed
1486
+ // here is immutable (duplicate ids are rejected), and while the
1487
+ // compile drops it, catching it now lets the model fix the target
1488
+ // in one receipt-free step.
1489
+ if (typeof rawEntry.symbol === "string" &&
1490
+ isSuspiciousVerificationSymbol(rawEntry.symbol)) {
1491
+ return planToolReceipt({
1492
+ ok: false,
1493
+ kind: "plan-verification-target",
1494
+ error: `record_plan_verification_target symbol "${rawEntry.symbol}" looks fabricated; use a real exported/describe/it symbol from ${rawEntry.file ?? "the target file"} or omit the symbol entirely (the trace gate verifies file+command)`,
1495
+ });
1496
+ }
1480
1497
  // uiStates: [] means this verification target is intentionally not
1481
1498
  // bound to a named UI state. Keep that canonical representation even
1482
1499
  // when a model omits the optional tool-boundary field.
@@ -1539,7 +1556,6 @@ export async function createFrontendPlanLedgerTools(input) {
1539
1556
  entry: evidenceGapSchema,
1540
1557
  }, { additionalProperties: false }),
1541
1558
  async execute(_toolCallId, params) {
1542
- await sleepRecordThrottle();
1543
1559
  const rawEntry = params?.entry;
1544
1560
  if (!isRecordObject(rawEntry)) {
1545
1561
  return planToolReceipt({
@@ -1732,6 +1748,19 @@ export async function createFrontendPlanLedgerTools(input) {
1732
1748
  const committed = readCommittedEvents(store, attemptId);
1733
1749
  await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"), committed);
1734
1750
  },
1751
+ committedFactCount: () => readCommittedEvents(store, attemptId).length,
1752
+ committedRequirementIds: () => {
1753
+ const ids = new Set();
1754
+ for (const event of readCommittedEvents(store, attemptId)) {
1755
+ const fact = event.fact;
1756
+ if (!fact || fact.kind !== "plan-requirement")
1757
+ continue;
1758
+ const entryFact = fact.entry;
1759
+ if (typeof entryFact?.id === "string")
1760
+ ids.add(entryFact.id);
1761
+ }
1762
+ return ids;
1763
+ },
1735
1764
  };
1736
1765
  }
1737
1766
  function isRecordObject(value) {
@@ -1869,7 +1898,6 @@ export async function createFrontendContractTools(input) {
1869
1898
  promptSnippet: `Commit 1-5 origin=contract ${kind} facts (up to 5 per message).`,
1870
1899
  parameters: Type.Object({}, { additionalProperties: true }),
1871
1900
  async execute(_toolCallId, params) {
1872
- await sleepRecordThrottle();
1873
1901
  const result = await adoptContractFact(kind, {
1874
1902
  kind,
1875
1903
  origin: "contract",
@@ -2524,6 +2552,171 @@ async function runFrontendDesignTerminalShadow(input) {
2524
2552
  }
2525
2553
  return input.mapped;
2526
2554
  }
2555
+ /** Tool subsets for the three frontend plan sessions (r17 split design). */
2556
+ const FRONTEND_PLAN_SEGMENTS = [
2557
+ {
2558
+ id: "coverage",
2559
+ toolNames: new Set([
2560
+ "record_plan_requirement",
2561
+ "record_plan_verification_target",
2562
+ "adopt_staged_fact",
2563
+ ]),
2564
+ instruction: [
2565
+ "PLAN SEGMENT 1/3 — coverage mapping only.",
2566
+ "Your ONLY job: for every frozen requirement, emit record_plan_requirement (requirement → implementation files) and record_plan_verification_target (verification target bound to requirement ids and files). Do NOT record components, UI states, mock, dependency, or routes — a follow-up session owns those.",
2567
+ "Do not call finalize_plan; it is not available in this segment.",
2568
+ ].join(" "),
2569
+ },
2570
+ {
2571
+ id: "ux-decisions",
2572
+ toolNames: new Set([
2573
+ "record_component_choice",
2574
+ "record_state_flow",
2575
+ "record_data_flow",
2576
+ "record_mock_api",
2577
+ "record_design_deviation",
2578
+ "record_dependency",
2579
+ "record_route_selection",
2580
+ "adopt_staged_fact",
2581
+ ]),
2582
+ instruction: [
2583
+ "PLAN SEGMENT 2/3 — UX decisions.",
2584
+ "Requirements and verification targets are already committed in the ledger (do NOT re-record them; duplicates are rejected). Your ONLY job: record component choices (decision=new requires sourceRequirementIds per the citation table), state flows, data flow, mock strategy, design deviation, dependency policy, and route selection.",
2585
+ "Do not call finalize_plan; it is not available in this segment.",
2586
+ ].join(" "),
2587
+ },
2588
+ {
2589
+ id: "finalize",
2590
+ toolNames: null,
2591
+ instruction: [
2592
+ "PLAN SEGMENT 3/3 — finalize.",
2593
+ "All record_* tools are available: if a finalize_plan receipt reports missing or invalid facts, fix them with the named record_* tool and call finalize_plan again. Otherwise call finalize_plan exactly once with no extra fields.",
2594
+ ].join(" "),
2595
+ },
2596
+ ];
2597
+ /**
2598
+ * Coverage-batch slicing for the frontend plan split (options 1+5): the
2599
+ * coverage segment becomes one session per requirement slice (default 4
2600
+ * requirements), so a small output budget can never be exhausted by
2601
+ * upfront reasoning about the whole requirement list. Zero-progress
2602
+ * batches are split in half and retried (option 5); single-requirement
2603
+ * zero-progress failures short-circuit to the retry ladder.
2604
+ */
2605
+ const FRONTEND_PLAN_COVERAGE_BATCH_SIZE = 4;
2606
+ const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 32;
2607
+ export async function runFrontendPlanSegmentedSessions(input) {
2608
+ const queue = [];
2609
+ // An empty list means "ledger unreadable / unknown" and must fall back to
2610
+ // one unscoped coverage session — only a non-empty list batches.
2611
+ const requirementIdsProvided = input.requirementIds !== undefined && input.requirementIds.length > 0;
2612
+ const pending = (input.requirementIds ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
2613
+ for (const segment of FRONTEND_PLAN_SEGMENTS) {
2614
+ if (segment.id !== "coverage") {
2615
+ queue.push({
2616
+ id: segment.id,
2617
+ toolNames: segment.toolNames,
2618
+ prompt: `${input.basePrompt}\n\n${segment.instruction}`,
2619
+ });
2620
+ continue;
2621
+ }
2622
+ // No requirement list (unreadable ledger) -> one unscoped coverage
2623
+ // session. A provided list with everything committed (resume) skips
2624
+ // coverage entirely.
2625
+ if (!requirementIdsProvided || pending.length > 0) {
2626
+ if (!requirementIdsProvided) {
2627
+ queue.push({
2628
+ id: segment.id,
2629
+ toolNames: segment.toolNames,
2630
+ prompt: `${input.basePrompt}\n\n${segment.instruction}`,
2631
+ });
2632
+ continue;
2633
+ }
2634
+ for (let i = 0; i < pending.length; i += FRONTEND_PLAN_COVERAGE_BATCH_SIZE) {
2635
+ const slice = pending.slice(i, i + FRONTEND_PLAN_COVERAGE_BATCH_SIZE);
2636
+ queue.push({
2637
+ id: `coverage-batch-${i / FRONTEND_PLAN_COVERAGE_BATCH_SIZE + 1}`,
2638
+ toolNames: segment.toolNames,
2639
+ coverageSlice: slice,
2640
+ prompt: `${input.basePrompt}\n\n${segment.instruction}\n\nCOVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them.`,
2641
+ });
2642
+ }
2643
+ }
2644
+ }
2645
+ let last;
2646
+ let index = 0;
2647
+ while (index < queue.length && index < FRONTEND_PLAN_BATCH_MAX_SESSIONS) {
2648
+ const session = queue[index];
2649
+ const remaining = (session.coverageSlice ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
2650
+ // Resume/earlier-batch commits may already cover this slice.
2651
+ if (session.coverageSlice && remaining.length === 0) {
2652
+ index += 1;
2653
+ continue;
2654
+ }
2655
+ let prompt = session.prompt;
2656
+ if (session.coverageSlice) {
2657
+ prompt = prompt.replace(/COVERAGE BATCH: process ONLY these requirements in this session: .*/, `COVERAGE BATCH: process ONLY these requirements in this session: ${remaining.join(", ")}. Other requirements are handled by separate sessions; do not record them.`);
2658
+ }
2659
+ const committedBefore = input.committedFactCount();
2660
+ const customTools = input.segmentCustomTools(session.toolNames);
2661
+ const result = await input.piStepFn({
2662
+ ...input.sessionOptions,
2663
+ prompt,
2664
+ ...(customTools.length > 0
2665
+ ? {
2666
+ writerToolPolicy: {
2667
+ requireSdk: true,
2668
+ customTools,
2669
+ },
2670
+ }
2671
+ : {}),
2672
+ });
2673
+ last = result;
2674
+ try {
2675
+ await input.flushLedger();
2676
+ }
2677
+ catch {
2678
+ // best-effort: the node-level flush runs again after the attempt
2679
+ }
2680
+ const committedAfter = input.committedFactCount();
2681
+ if (result.ok) {
2682
+ index += 1;
2683
+ continue;
2684
+ }
2685
+ const committedFactsOnlySuccess = session.id !== "finalize" &&
2686
+ !(result.assistantText ?? "").trim() &&
2687
+ !result.stderr.trim() &&
2688
+ !result.timedOut &&
2689
+ committedAfter > committedBefore;
2690
+ if (committedFactsOnlySuccess) {
2691
+ // The session died but banked facts: keep the progress and move on.
2692
+ index += 1;
2693
+ continue;
2694
+ }
2695
+ // Option 5: a multi-requirement coverage batch that failed with ZERO
2696
+ // new facts and no provider stderr is the upfront-reasoning burn —
2697
+ // halve the slice and retry instead of failing the attempt.
2698
+ const coverageSlice = session.coverageSlice;
2699
+ const zeroProgressBurn = coverageSlice !== undefined &&
2700
+ coverageSlice.length > 1 &&
2701
+ committedAfter === committedBefore &&
2702
+ !(result.assistantText ?? "").trim() &&
2703
+ !result.stderr.trim() &&
2704
+ !result.timedOut;
2705
+ if (zeroProgressBurn && coverageSlice) {
2706
+ const half = Math.ceil(coverageSlice.length / 2);
2707
+ queue.splice(index, 1, { ...session, coverageSlice: coverageSlice.slice(0, half) }, { ...session, coverageSlice: coverageSlice.slice(half) });
2708
+ continue;
2709
+ }
2710
+ return result;
2711
+ }
2712
+ return (last ?? {
2713
+ ok: false,
2714
+ stdout: "",
2715
+ stderr: "frontend plan segmentation produced no session",
2716
+ failureCategory: "empty-output",
2717
+ durationMs: 0,
2718
+ });
2719
+ }
2527
2720
  export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
2528
2721
  const started = Date.now();
2529
2722
  const persona = resolveDagPiPersona(input.task);
@@ -2906,10 +3099,9 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
2906
3099
  const piExtensionPaths = piExtensionsResolution && resolvePiBackend() !== "cli-only"
2907
3100
  ? piExtensionsResolution.resolved.flatMap((entry) => entry.entryPaths)
2908
3101
  : undefined;
2909
- result = await piStepFn({
3102
+ const piSessionOptions = {
2910
3103
  attachedFiles: [],
2911
3104
  modelConfig: resolveDagPiModelConfig(input.model, input.thinking ? { thinking: input.thinking } : undefined),
2912
- prompt: input.prompt,
2913
3105
  repoRoot: input.cwd,
2914
3106
  step,
2915
3107
  toolNames: resolveDagPiToolNames(input.task),
@@ -2922,7 +3114,6 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
2922
3114
  ...(input.task.contextBudget
2923
3115
  ? { contextBudget: input.task.contextBudget }
2924
3116
  : {}),
2925
- ...(writerToolPolicy ? { writerToolPolicy } : {}),
2926
3117
  ...(piExtensionPaths && piExtensionPaths.length > 0
2927
3118
  ? { piExtensionPaths }
2928
3119
  : {}),
@@ -2935,7 +3126,55 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
2935
3126
  bridgeActivity(activity.kind, activity.at);
2936
3127
  }
2937
3128
  : undefined,
2938
- });
3129
+ };
3130
+ if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
3131
+ // Frontend-only split: three sequential sessions with independent
3132
+ // output budgets (coverage -> UX decisions -> finalize), mirroring the
3133
+ // backend-test template's module sharding. Every other template keeps
3134
+ // the single-session path below.
3135
+ let planRequirementIds = [];
3136
+ try {
3137
+ const { readCommittedOriginFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
3138
+ const contractFacts = await readCommittedOriginFacts(meta.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl");
3139
+ // Only requirement facts: the contract ledger also carries
3140
+ // constraints (CON-*), evidence expectations (EV-*), handoff
3141
+ // intents (HND-*), open questions (OQ-*) and split proposals
3142
+ // (SPLIT-*) that all have ids — feeding those into the coverage
3143
+ // batches made the model record non-frozen plan-requirement ids
3144
+ // that finalize's canonical-coverage gate then rejected (r-ext2).
3145
+ planRequirementIds = contractFacts
3146
+ .filter((record) => record.fact
3147
+ ?.kind === "requirement")
3148
+ .map((record) => record.fact?.id)
3149
+ .filter((id) => typeof id === "string")
3150
+ .sort();
3151
+ }
3152
+ catch {
3153
+ // Unreadable ledger falls back to a single coverage session.
3154
+ }
3155
+ result = await runFrontendPlanSegmentedSessions({
3156
+ piStepFn,
3157
+ sessionOptions: piSessionOptions,
3158
+ basePrompt: input.prompt,
3159
+ attempt: input.attempt ?? 1,
3160
+ committedFactCount: () => planLedgerTools.committedFactCount(),
3161
+ requirementIds: planRequirementIds,
3162
+ committedRequirementIds: () => planLedgerTools.committedRequirementIds(),
3163
+ segmentCustomTools: (toolNames) => toolNames === null
3164
+ ? planLedgerTools.customTools
3165
+ : planLedgerTools.customTools.filter((tool) => typeof tool === "object" &&
3166
+ tool !== null &&
3167
+ toolNames.has(tool.name)),
3168
+ flushLedger: () => planLedgerTools.flush(),
3169
+ });
3170
+ }
3171
+ else {
3172
+ result = await piStepFn({
3173
+ ...piSessionOptions,
3174
+ prompt: input.prompt,
3175
+ ...(writerToolPolicy ? { writerToolPolicy } : {}),
3176
+ });
3177
+ }
2939
3178
  }
2940
3179
  catch (error) {
2941
3180
  if (playwrightToolContext) {
@@ -3072,25 +3311,10 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
3072
3311
  catch {
3073
3312
  // best-effort flush; missing ledger still fails at the node validator
3074
3313
  }
3075
- // Read-burst guard: a plan that burned dozens of read/grep/ls/find
3076
- // calls (re-reading upstream outputs and source files it should trust
3077
- // from typed facts) blows up the context window and eventually fails
3078
- // with 400 request-too-large. Detect it deterministically from the
3079
- // session log and retry with a reduced-reading instruction.
3080
- const readBurstIssues = input.task.readBudget?.onExhaustion === "return-guidance"
3081
- ? []
3082
- : await detectPlanReadBurst({
3083
- runDir: meta.runDir,
3084
- nodeId: input.task.id,
3085
- });
3086
- if (readBurstIssues.length > 0) {
3087
- return {
3088
- ...mapped,
3089
- ok: false,
3090
- failureCategory: "read-burst",
3091
- stderr: [mapped.stderr, ...readBurstIssues].filter(Boolean).join("\n\n"),
3092
- };
3093
- }
3314
+ // NOTE: the legacy plan read-burst guard lived here. It is dead code
3315
+ // since the plan node went tool-only (resolveDagPiToolNames grants no
3316
+ // read/grep/ls/find), so a read burst is structurally impossible; the
3317
+ // read-budget path (scout etc.) keeps its own guard.
3094
3318
  }
3095
3319
  if (!isWriteTask) {
3096
3320
  if (!mapped.ok &&
@@ -90,6 +90,10 @@ export async function adoptTaskContract(input) {
90
90
  requirementBytes: bytes.requirementBytes,
91
91
  constraintsBytes: bytes.constraintsBytes,
92
92
  referenceManifestBytes: bytes.referenceManifestBytes,
93
+ ...(state.ref?.sourceHashes.requirementLedgerSha256 &&
94
+ bytes.requirementLedgerBytes !== null
95
+ ? { requirementLedgerBytes: bytes.requirementLedgerBytes }
96
+ : {}),
93
97
  });
94
98
  const taskConfigSha256 = computeTaskConfigSha256(fullConfig);
95
99
  const canonicalHash = computeCanonicalHash({
@@ -45,6 +45,10 @@ export async function refreshManagedContractAfterImport(input) {
45
45
  requirementBytes: bytes.requirementBytes,
46
46
  constraintsBytes: bytes.constraintsBytes,
47
47
  referenceManifestBytes: bytes.referenceManifestBytes,
48
+ ...(state2.ref.sourceHashes.requirementLedgerSha256 &&
49
+ bytes.requirementLedgerBytes !== null
50
+ ? { requirementLedgerBytes: bytes.requirementLedgerBytes }
51
+ : {}),
48
52
  });
49
53
  const taskConfigSha256 = computeTaskConfigSha256(fullConfig);
50
54
  const canonicalHash = computeCanonicalHash({
@@ -539,6 +539,18 @@ export const frontendImplementationContractSchema = z
539
539
  message: `unknown verification target ${targetId}`,
540
540
  path: ["requirements"],
541
541
  });
542
+ // Interactions bind to VTs too: an interaction whose behavior
543
+ // verification points at an unmaterialized VT is untraceable —
544
+ // r20 review finding (dangling *-BEHAVIOR references sailed
545
+ // through plan/design/implement to the final review).
546
+ for (const interaction of value.interactions)
547
+ for (const targetId of interaction.verificationTargetIds)
548
+ if (!verificationIds.includes(targetId))
549
+ ctx.addIssue({
550
+ code: "custom",
551
+ message: `interaction "${interaction.name}" references unknown verification target ${targetId}`,
552
+ path: ["interactions"],
553
+ });
542
554
  // Source fidelity ledger binding (AC-005/AC-006): 绑定携带
543
555
  // requirementToFragments(ledger v2)时,每个 requirement 必须携带
544
556
  // 非空 sourceFragmentIds,否则 fail closed(防引用伪造/缺失)。
@@ -570,12 +582,23 @@ export const frontendImplementationContractSchema = z
570
582
  if (state.applicable &&
571
583
  (!state.expectedBehavior ||
572
584
  !state.implementationTargets?.length ||
573
- !state.verificationTargetIds?.length))
585
+ !state.verificationTargetIds?.length)) {
586
+ // Name the state and the exact missing fields: the fixer is a
587
+ // model iterating on finalize receipts — it cannot fix a
588
+ // defect it cannot locate (r18: 39 blind finalize retries).
589
+ const missing = [];
590
+ if (!state.expectedBehavior)
591
+ missing.push("expectedBehavior");
592
+ if (!state.implementationTargets?.length)
593
+ missing.push("implementationTargets");
594
+ if (!state.verificationTargetIds?.length)
595
+ missing.push("verificationTargetIds");
574
596
  ctx.addIssue({
575
597
  code: "custom",
576
- message: "applicable UI state requires behavior, implementation, and verification",
598
+ message: `applicable UI state "${state.name}" is missing: ${missing.join(", ")} — record_state_flow it again with those fields filled`,
577
599
  path: ["uiStates"],
578
600
  });
601
+ }
579
602
  if (!state.applicable && !state.notApplicableReason)
580
603
  ctx.addIssue({
581
604
  code: "custom",
@@ -1826,12 +1849,25 @@ export async function analyzeFrontendImplementationContract(input) {
1826
1849
  if (parsedTargetFiles.some((file) => file.startsWith("/") || file.includes("\\")))
1827
1850
  fail("blocked", "invalid-output: frontend contract target paths must be relative POSIX paths", candidateJsonSha256);
1828
1851
  const parsedStates = asRecord(parsed)?.uiStates;
1829
- if (Array.isArray(parsedStates) && parsedStates.some((item) => {
1830
- const state = asRecord(item);
1831
- return state?.applicable === true &&
1832
- (!asString(state.expectedBehavior) || asStringArray(state.implementationTargets).length === 0 || asStringArray(state.verificationTargetIds).length === 0);
1833
- }))
1834
- fail("retryable-invalid", "invalid-output: applicable UI state requires behavior, implementation, and verification", candidateJsonSha256);
1852
+ if (Array.isArray(parsedStates)) {
1853
+ const incompleteStates = [];
1854
+ for (const item of parsedStates) {
1855
+ const state = asRecord(item);
1856
+ if (!state || state.applicable !== true)
1857
+ continue;
1858
+ const missing = [];
1859
+ if (!asString(state.expectedBehavior))
1860
+ missing.push("expectedBehavior");
1861
+ if (asStringArray(state.implementationTargets).length === 0)
1862
+ missing.push("implementationTargets");
1863
+ if (asStringArray(state.verificationTargetIds).length === 0)
1864
+ missing.push("verificationTargetIds");
1865
+ if (missing.length > 0)
1866
+ incompleteStates.push(`"${asString(state.name)}": missing ${missing.join(", ")}`);
1867
+ }
1868
+ if (incompleteStates.length > 0)
1869
+ fail("retryable-invalid", `invalid-output: applicable UI states incomplete — re-record each with record_state_flow filling the named fields: ${incompleteStates.join("; ")}`, candidateJsonSha256);
1870
+ }
1835
1871
  const parsedMockApi = asRecord(parsed)?.mockApi;
1836
1872
  if (asRecord(parsedMockApi) &&
1837
1873
  typeof asRecord(parsedMockApi)?.strategy === "string" &&
@@ -2079,7 +2115,7 @@ const VERIFICATION_SYMBOL_MAX_CHARS = 60;
2079
2115
  * loading") and `describe(...)` / `it(...)` forms stay valid — a real symbol
2080
2116
  * may be a function name, a dotted path, or a describe/it title.
2081
2117
  */
2082
- function isSuspiciousVerificationSymbol(symbol) {
2118
+ export function isSuspiciousVerificationSymbol(symbol) {
2083
2119
  const trimmed = symbol.trim();
2084
2120
  if (!trimmed)
2085
2121
  return false;
@@ -2126,6 +2162,17 @@ export class PlanPolicyPrecheckFailure extends Error {
2126
2162
  */
2127
2163
  export async function analyzeFrontendPlanPatchCandidate(input) {
2128
2164
  const analysis = await analyzeFrontendImplementationContract(input);
2165
+ // A committed VT with a fabricated symbol is an immutable ledger fact —
2166
+ // the record boundary rejects duplicate ids, so the model cannot overwrite
2167
+ // it and throwing here deadlocks the receipt loop (r19: VT-AC006-BEHAVIOR).
2168
+ // Drop suspicious symbols deterministically instead: the VT stays valid and
2169
+ // the deterministic trace gate verifies file+command (and resolvability)
2170
+ // after verification. Mirrors the shell materialization's drop semantics.
2171
+ for (const target of analysis.canonical.verificationTargets) {
2172
+ if (target.symbol && isSuspiciousVerificationSymbol(target.symbol)) {
2173
+ target.symbol = undefined;
2174
+ }
2175
+ }
2129
2176
  // Front-load the verification-symbol shape check so fabricated symbols
2130
2177
  // are fixed by the plan retry ladder in-node instead of failing the
2131
2178
  // verify trace gate at the end of the run.