@tea-agent/loop-agent 0.39.0-beta.13 → 0.39.0-beta.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,6 +1,7 @@
1
1
  # 更新日志
2
2
 
3
3
  ## [Unreleased]
4
+ - 0.39.0-beta.14(frontend-node dogfood):plan-pi 三段分段会话 + coverage 需求批次切片(4/批、零进度减半)、finalize 收据修复环(聚合诊断+修复 JSON)、record_* 边界校验左移(VT 重复/writeSet/交互 VT 引用/规范需求 id/可疑 symbol)、编译装配 last-wins(组件按 purpose 去重修正)、review-context 授权面扩展至 VT 测试文件、设计评审修订降级为 admission advisory。
4
5
  - 前端 DAG 极端环境加固(16KB 输出 / 120KB 上下文真实冒烟,r2-r5 验证):`frontend-contract-pi` 改为消费编译后的 ledger 输入块(`<frontend_contract_input>`,12KB 上限、cap ladder、requirement id 永不丢)并移除 read 工具,以增量 `record_*` 提交为唯一输出模式(OUTPUT BUDGET DISCIPLINE),修复极端小输出窗口下模型反复 read 源文件空转 1h54m 超时的问题(r3 起 contract 3m54s 正常完成、37 次 record_* / 0 次 read);plan/contract 的 `record_*` 工具支持每 message 最多 5 条批量提交并加 1s 调用间隔,rate-limit 退避上限提高到 60s,缓解密集小调用触发第三方网关限流;修复 fragment-inventory 对 V4 标题式 AC(`#### AC-NNN`)的双重 emit 缺陷(同一标题行被 section 入口与正文 heading 分支各切一次,30 个 AC 结构性产生 30 个 duplicate)。
5
6
 
6
7
  - 修复连续 `dag rerun-task` 丢失前代评审意见的问题:直接父 run 未走到 review/design review 时,`deriveDagRerunFeedback` 沿 run-owned lineage 有界追溯最近 5 层并继承最近的 committed typed findings;较近的 `approve_review` / `approve_design` 作为对应 resolution barrier,防止已解决 finding 被复活。lineage run ID、生命周期目录与 `state.json.runId` 必须一致,损坏、越界、循环或超深链路 fail closed。
@@ -260,6 +260,10 @@ async function monitorRun(input) {
260
260
  failureCategory: "monitor-timeout",
261
261
  });
262
262
  }
263
+ function isStartupGateEligible(snapshot) {
264
+ const status = snapshot.source?.sourceFidelityStartup?.status;
265
+ return status === undefined || status === "complete" || status === "not-applicable";
266
+ }
263
267
  function nextMutationAction(actions) {
264
268
  for (const action of actions) {
265
269
  if (action.type !== "stop")
@@ -379,7 +383,10 @@ export async function advanceTaskLifecycle(input) {
379
383
  };
380
384
  }
381
385
  const token = parseGateToken(input.rejectGateToken);
382
- const currentGate = snapshot.gate ?? (await buildCurrentGate(input.repoRoot, input.taskId));
386
+ const currentGate = snapshot.gate ??
387
+ (isStartupGateEligible(snapshot)
388
+ ? await buildCurrentGate(input.repoRoot, input.taskId)
389
+ : null);
383
390
  if (!currentGate || !gateMatches(currentGate, token)) {
384
391
  return {
385
392
  taskId: input.taskId,
@@ -443,11 +450,12 @@ export async function advanceTaskLifecycle(input) {
443
450
  dryRun,
444
451
  skipFinalize: Boolean(input.skipFinalize),
445
452
  forcePrepareContract: Boolean(input.fromDraftPath) && !preparedContractThisCall,
453
+ startupRepairAttempted: preparedContractThisCall,
446
454
  });
447
455
  if (dryRun) {
448
456
  const plan = planOnce(snapshot);
449
457
  const plannedGate = snapshot.gate ??
450
- (snapshot.dagDraft?.exists
458
+ (isStartupGateEligible(snapshot) && snapshot.dagDraft?.exists
451
459
  ? await buildCurrentGate(input.repoRoot, input.taskId)
452
460
  : null);
453
461
  return {
@@ -586,6 +594,10 @@ export async function advanceTaskLifecycle(input) {
586
594
  continue;
587
595
  }
588
596
  if (action.type === "prepare-contract") {
597
+ const startupStatus = snapshot.source?.sourceFidelityStartup?.status;
598
+ const startupRepairAction = startupStatus === "missing" ||
599
+ startupStatus === "invalid" ||
600
+ startupStatus === "coverage-incomplete";
589
601
  // 生产 source fidelity Pi 接线(AC-PROD-001/005):仅 frontend-implementation
590
602
  // 任务注入 reconcile executor + 独立 reviewer;ineligible 返回 undefined 不接线。
591
603
  // 主路径与 semantic-intake retry 路径共用同一接线,保证任意一条生产路径都实际
@@ -731,7 +743,9 @@ export async function advanceTaskLifecycle(input) {
731
743
  message: "structured requirement draft via semantic intake; projected managed source",
732
744
  });
733
745
  preparedContractThisCall = true;
734
- if (refreshManagedContractThisCall || input.fromDraftPath) {
746
+ if (startupRepairAction ||
747
+ refreshManagedContractThisCall ||
748
+ input.fromDraftPath) {
735
749
  refreshManagedContractThisCall = false;
736
750
  regenerateDagThisCall = true;
737
751
  }
@@ -825,7 +839,9 @@ export async function advanceTaskLifecycle(input) {
825
839
  break;
826
840
  }
827
841
  preparedContractThisCall = true;
828
- if (refreshManagedContractThisCall || input.fromDraftPath) {
842
+ if (startupRepairAction ||
843
+ refreshManagedContractThisCall ||
844
+ input.fromDraftPath) {
829
845
  // The old DAG and gate were bound to the pre-mutation contract.
830
846
  refreshManagedContractThisCall = false;
831
847
  regenerateDagThisCall = true;
@@ -1,9 +1,11 @@
1
- import { access, readFile } from "node:fs/promises";
1
+ import { access, readFile, readdir } from "node:fs/promises";
2
2
  import path from "node:path";
3
3
  import { getTaskDir, getTaskPaths, getTaskStatus, loadTaskConfig, } from "../../task/runtime.js";
4
4
  import { logicalRevisionForState, observeTaskContractState, } from "../../task/contract/index.js";
5
5
  import { assessDagRunLiveness, listAllDagRunEntries, readDagRunSpec, readDagRunState, } from "../../workflows/dag/lifecycle.js";
6
6
  import { resolveRunningControllerIdentity } from "../../shared/package-metadata.js";
7
+ import { parseLedgerJson, validateRequirementLedger, } from "../../task/source-prepare/ledger.js";
8
+ import { REQUIREMENT_LEDGER_FILE_NAME } from "../../task/source-prepare/types.js";
7
9
  import { loadLifecycleRecord } from "./record.js";
8
10
  import { buildGateBinding, buildWriteSetGate, buildWriteSetGateDelta, controllerIdentityAnchor, hashCanonicalPayload, normalizeWriteSet, sha256Hex, } from "./gates.js";
9
11
  async function exists(filePath) {
@@ -47,6 +49,140 @@ async function taskExists(repoRoot, taskId) {
47
49
  const paths = getTaskPaths(repoRoot, taskId);
48
50
  return exists(paths.taskConfigPath);
49
51
  }
52
+ async function assessReferenceEvidence(referencesDir) {
53
+ try {
54
+ const entries = await readdir(referencesDir);
55
+ const importedPrdCount = entries.filter((name) => name.endsWith(".md")).length;
56
+ return importedPrdCount > 0
57
+ ? { status: "present", importedPrdCount }
58
+ : { status: "absent-or-empty", importedPrdCount: 0 };
59
+ }
60
+ catch (error) {
61
+ if (error &&
62
+ typeof error === "object" &&
63
+ "code" in error &&
64
+ error.code === "ENOENT") {
65
+ return { status: "absent-or-empty", importedPrdCount: 0 };
66
+ }
67
+ return {
68
+ status: "unknown",
69
+ importedPrdCount: 0,
70
+ error: error instanceof Error ? error.message : String(error),
71
+ };
72
+ }
73
+ }
74
+ function hasLedgerStructure(value) {
75
+ if (!value || typeof value !== "object")
76
+ return false;
77
+ const ledger = value;
78
+ return (typeof ledger.taskId === "string" &&
79
+ typeof ledger.inputDigest === "string" &&
80
+ typeof ledger.requirementSha256 === "string" &&
81
+ Array.isArray(ledger.sourcePaths) &&
82
+ Array.isArray(ledger.fragments) &&
83
+ Array.isArray(ledger.canonicalRequirements) &&
84
+ Boolean(ledger.stats && typeof ledger.stats === "object") &&
85
+ Array.isArray(ledger.unresolved) &&
86
+ Array.isArray(ledger.conflicts) &&
87
+ Array.isArray(ledger.sourceResolutions) &&
88
+ typeof ledger.schemaReconcilerVersion === "string" &&
89
+ typeof ledger.builtAt === "string");
90
+ }
91
+ async function assessSourceFidelityStartup(input) {
92
+ if (!input.applicable) {
93
+ return {
94
+ status: "not-applicable",
95
+ ledgerPath: input.ledgerPath,
96
+ diagnostics: [],
97
+ };
98
+ }
99
+ if (input.referenceEvidenceError) {
100
+ return {
101
+ status: "invalid",
102
+ ledgerPath: input.ledgerPath,
103
+ diagnostics: [
104
+ {
105
+ code: "SOURCE_FIDELITY_LEDGER_STALE",
106
+ message: `imported PRD reference evidence is unreadable: ${input.referenceEvidenceError}`,
107
+ },
108
+ ],
109
+ };
110
+ }
111
+ let raw;
112
+ try {
113
+ raw = await readFile(input.ledgerPath, "utf-8");
114
+ }
115
+ catch (error) {
116
+ const missing = !(await exists(input.ledgerPath));
117
+ return {
118
+ status: missing ? "missing" : "invalid",
119
+ ledgerPath: input.ledgerPath,
120
+ diagnostics: [
121
+ {
122
+ code: missing
123
+ ? "SOURCE_FIDELITY_LEDGER_MISSING"
124
+ : "SOURCE_FIDELITY_LEDGER_STALE",
125
+ message: missing
126
+ ? "managed frontend task has no requirement-ledger.json"
127
+ : `requirement ledger is unreadable: ${error instanceof Error ? error.message : String(error)}`,
128
+ },
129
+ ],
130
+ };
131
+ }
132
+ try {
133
+ const ledger = parseLedgerJson(raw);
134
+ if (!hasLedgerStructure(ledger)) {
135
+ return {
136
+ status: "invalid",
137
+ ledgerPath: input.ledgerPath,
138
+ diagnostics: [
139
+ {
140
+ code: "SOURCE_FIDELITY_LEDGER_STALE",
141
+ message: "requirement ledger is missing required structural fields",
142
+ },
143
+ ],
144
+ };
145
+ }
146
+ const nonDecisionErrors = validateRequirementLedger(ledger, {
147
+ allowedUnresolved: true,
148
+ allowedConflicts: true,
149
+ });
150
+ if (nonDecisionErrors.length > 0) {
151
+ const coverageCodes = new Set([
152
+ "SOURCE_FIDELITY_HIGH_RISK_UNCOVERED",
153
+ "SOURCE_FIDELITY_REQUIREMENT_NOT_COVERED",
154
+ ]);
155
+ return {
156
+ status: nonDecisionErrors.every((error) => coverageCodes.has(error.code))
157
+ ? "coverage-incomplete"
158
+ : "invalid",
159
+ ledgerPath: input.ledgerPath,
160
+ diagnostics: nonDecisionErrors,
161
+ };
162
+ }
163
+ const decisionErrors = validateRequirementLedger(ledger);
164
+ if (decisionErrors.length > 0) {
165
+ return {
166
+ status: "human-decision-blocked",
167
+ ledgerPath: input.ledgerPath,
168
+ diagnostics: decisionErrors,
169
+ };
170
+ }
171
+ return { status: "complete", ledgerPath: input.ledgerPath, diagnostics: [] };
172
+ }
173
+ catch (error) {
174
+ return {
175
+ status: "invalid",
176
+ ledgerPath: input.ledgerPath,
177
+ diagnostics: [
178
+ {
179
+ code: "SOURCE_FIDELITY_LEDGER_STALE",
180
+ message: error instanceof Error ? error.message : String(error),
181
+ },
182
+ ],
183
+ };
184
+ }
185
+ }
50
186
  /**
51
187
  * Explicit Task↔DAG association only:
52
188
  * - lifecycle.json recorded runAssociations
@@ -182,10 +318,12 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
182
318
  const record = await loadLifecycleRecord(repoRoot, taskId);
183
319
  const contractState = await observeTaskContractState(repoRoot, taskId);
184
320
  let title;
321
+ let taskKind;
185
322
  let taskConfigAllowed = [];
186
323
  try {
187
324
  const config = await loadTaskConfig(repoRoot, taskId);
188
325
  title = config.title;
326
+ taskKind = config.taskKind;
189
327
  taskConfigAllowed = config.allowedPaths ?? [];
190
328
  }
191
329
  catch {
@@ -199,19 +337,22 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
199
337
  ...(requirementReady ? [] : [requirementPath]),
200
338
  ...(constraintsReady ? [] : [constraintsPath]),
201
339
  ];
202
- let importedPrdCount = 0;
340
+ const contractManaged = contractState.effectiveStatus === "managed" || Boolean(contractState.ref);
203
341
  const referencesDir = path.join(paths.sourceDir, "references");
204
- if (await exists(referencesDir)) {
205
- // Count is best-effort; absence is fine.
206
- try {
207
- const { readdir } = await import("node:fs/promises");
208
- const entries = await readdir(referencesDir);
209
- importedPrdCount = entries.filter((name) => name.endsWith(".md")).length;
210
- }
211
- catch {
212
- importedPrdCount = 0;
213
- }
214
- }
342
+ const referenceEvidence = await assessReferenceEvidence(referencesDir);
343
+ const importedPrdCount = referenceEvidence.importedPrdCount;
344
+ const ledgerPath = path.join(paths.sourceDir, REQUIREMENT_LEDGER_FILE_NAME);
345
+ const sourceFidelityStartup = await assessSourceFidelityStartup({
346
+ ledgerPath,
347
+ applicable: contractManaged &&
348
+ taskKind === "frontend-implementation" &&
349
+ referenceEvidence.status !== "absent-or-empty",
350
+ referenceEvidenceError: referenceEvidence.status === "unknown"
351
+ ? referenceEvidence.error
352
+ : undefined,
353
+ });
354
+ const sourceFidelityReady = sourceFidelityStartup.status === "complete" ||
355
+ sourceFidelityStartup.status === "not-applicable";
215
356
  const dagDraftPath = paths.dagDraftPath;
216
357
  const dagDraftExists = await exists(dagDraftPath);
217
358
  let specHash = null;
@@ -285,7 +426,11 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
285
426
  // Open gate only when draft is strict-validated and the current binding is
286
427
  // not already approved/rejected. Rejected same-digest gates stay closed
287
428
  // until contract/source/DAG/writeSet/controller binding changes.
288
- const gate = gateBound && strictValidated && !approvedCurrent && !rejectedCurrent
429
+ const gate = gateBound &&
430
+ strictValidated &&
431
+ sourceFidelityReady &&
432
+ !approvedCurrent &&
433
+ !rejectedCurrent
289
434
  ? buildWriteSetGate({ taskId, bound: gateBound, delta: gateDelta })
290
435
  : null;
291
436
  const gateOpen = Boolean(gate);
@@ -310,11 +455,10 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
310
455
  activeRun.failureCategory === "needs-attention" ||
311
456
  activeRun.failureCategory === "suspected-stall" ||
312
457
  activeRun.failureCategory === "monitor-timeout"));
313
- const contractManaged = contractState.effectiveStatus === "managed" || Boolean(contractState.ref);
314
458
  const lifecycleState = deriveLifecycleState({
315
459
  exists: true,
316
460
  contractManaged,
317
- sourceReady: requirementReady,
461
+ sourceReady: requirementReady && sourceFidelityReady,
318
462
  dagValidated: strictValidated,
319
463
  gateOpen: Boolean(gate),
320
464
  activeRun,
@@ -352,6 +496,7 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
352
496
  constraintsReady,
353
497
  importedPrdCount,
354
498
  missing,
499
+ sourceFidelityStartup,
355
500
  },
356
501
  dagDraft: {
357
502
  path: dagDraftPath,
@@ -375,7 +520,16 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
375
520
  exists: closeoutExists || Boolean(record?.closeout),
376
521
  },
377
522
  completedTransitions: record?.completedTransitions ?? [],
378
- blockers: [],
523
+ blockers: sourceFidelityReady
524
+ ? []
525
+ : sourceFidelityStartup.diagnostics.map((diagnostic) => ({
526
+ code: diagnostic.code,
527
+ message: diagnostic.message,
528
+ details: {
529
+ startupStatus: sourceFidelityStartup.status,
530
+ ledgerPath: sourceFidelityStartup.ledgerPath,
531
+ },
532
+ })),
379
533
  warnings: [],
380
534
  next,
381
535
  };
@@ -46,6 +46,37 @@ export function planTransitions(input) {
46
46
  actions.push({ type: "recover-contract" });
47
47
  expectedTransitions.push("contract-recovered");
48
48
  }
49
+ const startupStatus = snapshot.source?.sourceFidelityStartup?.status;
50
+ const startupRepairDetected = startupStatus === "missing" ||
51
+ startupStatus === "invalid" ||
52
+ startupStatus === "coverage-incomplete";
53
+ if (startupRepairDetected && input.startupRepairAttempted) {
54
+ actions.push({ type: "stop", reason: "blocker" });
55
+ return {
56
+ actions,
57
+ gate: null,
58
+ blockers,
59
+ next: {
60
+ kind: "advance",
61
+ command: `loop-agent task advance ${snapshot.taskId} --json`,
62
+ },
63
+ expectedTransitions,
64
+ };
65
+ }
66
+ const startupRepairRequired = startupRepairDetected;
67
+ if (startupStatus === "human-decision-blocked") {
68
+ actions.push({ type: "stop", reason: "blocker" });
69
+ return {
70
+ actions,
71
+ gate: null,
72
+ blockers,
73
+ next: {
74
+ kind: "external-action",
75
+ description: "Requirement ledger requires an existing human source decision; preserve sourceResolutions, resolve the blocker, then advance again.",
76
+ },
77
+ expectedTransitions,
78
+ };
79
+ }
49
80
  const sourceReady = snapshot.source?.requirementReady === true;
50
81
  const contractManaged = snapshot.contract?.effectiveStatus === "managed" ||
51
82
  Boolean(snapshot.contract?.revision && snapshot.contract.revision > 0);
@@ -53,7 +84,9 @@ export function planTransitions(input) {
53
84
  // Prefer prepare once PRDs are on disk; re-import only when caller still
54
85
  // signals hasPrdInput (advance.ts latches this after one import per call).
55
86
  const shouldImportPrd = input.hasPrdInput;
56
- if (input.forcePrepareContract || input.refreshManagedContract) {
87
+ if (input.forcePrepareContract ||
88
+ input.refreshManagedContract ||
89
+ startupRepairRequired) {
57
90
  if (shouldImportPrd) {
58
91
  actions.push({ type: "import-prd" });
59
92
  expectedTransitions.push("prd-imported");
@@ -124,7 +157,8 @@ export function planTransitions(input) {
124
157
  const strictValidated = snapshot.dagDraft?.strictValidated === true;
125
158
  const dagRefreshRequired = Boolean(input.forceDagRefresh ||
126
159
  input.refreshManagedContract ||
127
- input.forcePrepareContract);
160
+ input.forcePrepareContract ||
161
+ startupRepairRequired);
128
162
  if (!dagExists || dagRefreshRequired) {
129
163
  actions.push({ type: "generate-dag" });
130
164
  expectedTransitions.push("dag-generated");
@@ -145,10 +179,11 @@ export function planTransitions(input) {
145
179
  snapshot.lifecycleState === "evidence-closed" ||
146
180
  snapshot.lifecycleState === "needs-attention" ||
147
181
  snapshot.lifecycleState === "awaiting-decision";
148
- const approvedForCurrent = postRunState || (input.approveGate && gate !== null);
182
+ const approveCurrentGate = input.approveGate && !dagRefreshRequired;
183
+ const approvedForCurrent = postRunState || (approveCurrentGate && gate !== null);
149
184
  if (!approvedForCurrent && (strictValidated || !dagExists || dagRefreshRequired)) {
150
185
  // Gate opens after generate+validate in the advance loop; plan includes open.
151
- if (!input.approveGate) {
186
+ if (!approveCurrentGate) {
152
187
  actions.push({ type: "open-write-set-gate" });
153
188
  expectedTransitions.push("write-set-gate-opened");
154
189
  if (input.dryRun) {
@@ -177,14 +212,14 @@ export function planTransitions(input) {
177
212
  (snapshot.lifecycleState === "run-failed" && !failedRecoverable) ||
178
213
  snapshot.lifecycleState === "evidence-closed";
179
214
  if (!alreadyTerminalRun) {
180
- if (input.approveGate && !postRunState) {
215
+ if (approveCurrentGate && !postRunState) {
181
216
  expectedTransitions.push("write-set-approved");
182
217
  actions.push({ type: "start-or-monitor-run" });
183
218
  expectedTransitions.push("dag-run-started", "dag-run-monitored");
184
219
  }
185
220
  else if (snapshot.lifecycleState === "running" ||
186
221
  snapshot.activeRun ||
187
- (input.approveGate && Boolean(snapshot.activeRun))) {
222
+ (approveCurrentGate && Boolean(snapshot.activeRun))) {
188
223
  actions.push({ type: "start-or-monitor-run" });
189
224
  expectedTransitions.push("dag-run-monitored");
190
225
  }
@@ -218,7 +253,7 @@ export function planTransitions(input) {
218
253
  actions,
219
254
  gate,
220
255
  blockers,
221
- next: deriveNext(snapshot, gate, input.approveGate),
256
+ next: deriveNext(snapshot, gate, approveCurrentGate),
222
257
  expectedTransitions,
223
258
  };
224
259
  }
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "schemaVersion": 1,
3
- "version": "0.39.0-beta.13",
4
- "gitSha": "5d72ad152e1f52f7001782af509a53f11c330aa4",
5
- "builtAt": "2026-08-29T13:28:15.172Z"
3
+ "version": "0.39.0-beta.14",
4
+ "gitSha": "9d8bebca76a2608d9b38a1af1f379fd5fe5b04b0",
5
+ "builtAt": "2026-08-30T09:13:51.707Z"
6
6
  }
@@ -16,6 +16,7 @@ import { redactPromptForLog, truncateOutput, } from "../shared/output-truncation
16
16
  import { GitStatusUnavailableError, pathsChangedDuringRun, readGitStatusPorcelain, recoverRootNulArtifact, snapshotGitStatusPathFingerprints, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
17
17
  import { captureWorkspaceWriteSnapshot, diffWorkspaceWriteSnapshots, } from "./workspace-write-snapshot.js";
18
18
  import { pathMatchesPattern } from "../shared/git-progress.js";
19
+ import { isSuspiciousVerificationSymbol } from "../workflows/dag/frontend-implementation-contract.js";
19
20
  import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
20
21
  import { assessBackendTestMdPlanCompleteness, assessBackendTestMdWriterCompleteness, assessBackendTestPytestPlanCompleteness, assessBackendTestPytestWriterCompleteness, assessBackendTestShardChildCompleteness, backendTestWriterProgressRoleForTask, classifyBackendTestWriterCompletenessFailure, isBackendTestCompletenessRetryCandidate, isBackendTestMdPlanTask, isBackendTestPytestPlanTask, isBackendTestShardChildTask, writeBackendTestWriterProgressArtifacts, } from "../workflows/dag/backend-test-writer-completeness.js";
21
22
  import { resolveBackendTestLayout } from "../workflows/dag/backend-test-layout.js";
@@ -236,16 +237,6 @@ export const FRONTEND_PLAN_RECORD_TOOL_NAMES = [
236
237
  ];
237
238
  export const FRONTEND_PLAN_TERMINAL_TOOL_NAMES = ["finalize_plan"];
238
239
  export const FRONTEND_PLAN_ADOPT_TOOL_NAMES = ["adopt_staged_fact"];
239
- /**
240
- * Inter-call delay for high-frequency frontend record_* tools. Incremental
241
- * submission (one model response per 1-5 entries) means dozens of sequential
242
- * API round trips in a few minutes; third-party gateways (huu.dqy.ink) trip
243
- * rate limits whose windows exceed normal API RPM even at ~13 calls/min. A
244
- * short fixed delay before each tool body keeps the sustained rate under
245
- * typical thresholds while batching cuts the total call count.
246
- */
247
- const FRONTEND_RECORD_TOOL_THROTTLE_MS = 1000;
248
- const sleepRecordThrottle = () => new Promise((resolve) => setTimeout(resolve, FRONTEND_RECORD_TOOL_THROTTLE_MS));
249
240
  /** M5: `frontend-review-pi` emits its authoritative terminal verdict through
250
241
  * committed typed tools instead of the legacy JSON verdict parse. */
251
242
  export function isFrontendReviewTypedTerminalNode(task) {
@@ -395,10 +386,6 @@ export function scanReviewTerminalKindsFromSessionEvents(content) {
395
386
  }
396
387
  return kinds;
397
388
  }
398
- /** Read-only discovery tool budget for frontend-plan-pi. Exceeding it means
399
- * the plan re-read upstream outputs/sources instead of trusting typed facts,
400
- * which blows up the context window (400 request-too-large). */
401
- const PLAN_READ_TOOL_BUDGET = 40;
402
389
  const READ_ONLY_TOOL_NAMES = new Set(["read", "grep", "ls", "find"]);
403
390
  function readEventPath(event) {
404
391
  const candidates = [event.path, event.readPath, event.input];
@@ -483,45 +470,6 @@ export async function detectNodeReadBudget(input) {
483
470
  issues.push(`frontend ${input.nodeId} read budget exceeded: ${stats.elapsedMs}ms (budget ${input.budget.maxMs}ms)`);
484
471
  return issues;
485
472
  }
486
- /** Deterministic read-burst guard for frontend-plan-pi: count read-only
487
- * discovery tool calls (read/grep/ls/find) from the session log. Over budget →
488
- * read-burst, retried with a reduced-reading instruction. Pure scan; a
489
- * successful plan under budget is never blocked. */
490
- export async function detectPlanReadBurst(input) {
491
- const sessionEventsPath = path.join(input.runDir, input.nodeId, "session-events.jsonl");
492
- let count = 0;
493
- try {
494
- const content = await readFile(sessionEventsPath, "utf8");
495
- for (const line of content.split("\n")) {
496
- if (!line.trim())
497
- continue;
498
- try {
499
- const event = JSON.parse(line);
500
- if (event.type === "tool_execution_start" &&
501
- typeof event.toolName === "string" &&
502
- (event.toolName === "read" ||
503
- event.toolName === "grep" ||
504
- event.toolName === "ls" ||
505
- event.toolName === "find")) {
506
- count += 1;
507
- }
508
- }
509
- catch {
510
- // skip unparseable line
511
- }
512
- }
513
- }
514
- catch {
515
- // Missing/unreadable session log → no burst detection
516
- return [];
517
- }
518
- if (count > PLAN_READ_TOOL_BUDGET) {
519
- return [
520
- `frontend plan read-burst: ${count} read-only tool calls (budget ${PLAN_READ_TOOL_BUDGET}). Trust the upstream contract/scout typed facts; do not re-read contract/scout outputs or source files already captured. Minimize discovery reads, commit record_* facts directly, then finalize_plan.`,
521
- ];
522
- }
523
- return [];
524
- }
525
473
  export const FRONTEND_DESIGN_TERMINAL_TOOL_NAMES = new Set([
526
474
  "approve_design",
527
475
  "request_design_changes",
@@ -932,11 +880,27 @@ export async function createFrontendPlanLedgerTools(input) {
932
880
  import("typebox"),
933
881
  import("@earendil-works/pi-coding-agent"),
934
882
  ]);
935
- const { readCommittedEvents, writeTypedEventStoreJsonl } = await import("../workflows/dag/frontend-typed-event-store.js");
883
+ const { loadTypedEventStore, readCommittedEvents, writeTypedEventStoreJsonl, } = await import("../workflows/dag/frontend-typed-event-store.js");
936
884
  const { adoptStagedFact, adoptTypedEventFact, stageTypedEventFact } = await import("../workflows/dag/frontend-typed-event-transaction.js");
937
885
  const { assemblePlanPatchFromCommittedFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
938
886
  const store = input.store;
939
887
  const attemptId = input.attemptId;
888
+ // A retry creates a fresh executor-local store, but the plan ledger is the
889
+ // cross-attempt authority. Restore the committed prefix before registering
890
+ // tools; otherwise the first flush of a retry can overwrite facts that the
891
+ // previous attempt had already committed. The on-disk file contains only
892
+ // committed records, so loading it is also fail-closed with respect to
893
+ // staged/quarantined facts.
894
+ const persisted = await loadTypedEventStore(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"));
895
+ if (persisted.records.length > 0) {
896
+ const existingEventIds = new Set(store.records.map((record) => record.eventId));
897
+ for (const record of persisted.records) {
898
+ if (!existingEventIds.has(record.eventId)) {
899
+ store.records.push(record);
900
+ }
901
+ }
902
+ store.revision = Math.max(store.revision, persisted.revision);
903
+ }
940
904
  const stringArray = Type.Array(Type.String({}));
941
905
  const optionalString = Type.Optional(Type.String({}));
942
906
  const optionalStringArray = Type.Optional(stringArray);
@@ -1127,7 +1091,6 @@ export async function createFrontendPlanLedgerTools(input) {
1127
1091
  stylingStrategy: optionalString,
1128
1092
  }, { additionalProperties: false }),
1129
1093
  async execute(_toolCallId, params) {
1130
- await sleepRecordThrottle();
1131
1094
  const rawChoice = params?.choice;
1132
1095
  if (!isRecordObject(rawChoice)) {
1133
1096
  return planToolReceipt({
@@ -1180,7 +1143,16 @@ export async function createFrontendPlanLedgerTools(input) {
1180
1143
  ? { stylingStrategy: params.stylingStrategy }
1181
1144
  : {}),
1182
1145
  });
1183
- return planToolReceipt(result);
1146
+ // Echo the frozen citation the runtime derived: the model sees the
1147
+ // purpose↔citation mapping it just committed and can re-record the
1148
+ // choice (last-wins per purpose at compile) when it mismatches.
1149
+ const echo = {
1150
+ ...result,
1151
+ ...(choice.specReference
1152
+ ? { derivedSpecReference: choice.specReference }
1153
+ : {}),
1154
+ };
1155
+ return planToolReceipt(echo);
1184
1156
  },
1185
1157
  });
1186
1158
  const recordStateFlowTool = defineTool({
@@ -1264,6 +1236,26 @@ export async function createFrontendPlanLedgerTools(input) {
1264
1236
  error: "record_state_flow requires non-empty interaction.name (id is accepted only as a legacy alias)",
1265
1237
  });
1266
1238
  }
1239
+ // Interaction -> VT forward references are legal only against
1240
+ // already-committed VT facts. In the segmented flow every VT
1241
+ // commits in the coverage segment before state flows run, so a
1242
+ // dangling reference here is a real defect (r20: *-BEHAVIOR
1243
+ // refs reached the final review untraceable).
1244
+ const interactionVtIds = stringList(interaction.verificationTargetIds);
1245
+ const committedVtIds = new Set(readCommittedEvents(store, attemptId)
1246
+ .map((event) => event.fact)
1247
+ .filter((fact) => fact.kind === "plan-verification-target")
1248
+ .map((fact) => fact.entry
1249
+ ?.id)
1250
+ .filter((id) => typeof id === "string"));
1251
+ const unknownVtIds = interactionVtIds.filter((id) => !committedVtIds.has(id));
1252
+ if (unknownVtIds.length > 0) {
1253
+ return planToolReceipt({
1254
+ ok: false,
1255
+ kind: "state-flow",
1256
+ error: `record_state_flow interaction "${resolvedName}" references verification targets that are not recorded yet: ${unknownVtIds.join(", ")}; record them with record_plan_verification_target first, then re-record this state flow`,
1257
+ });
1258
+ }
1267
1259
  interactions.push({ ...interaction, name: resolvedName });
1268
1260
  }
1269
1261
  const states = uiStates
@@ -1359,7 +1351,6 @@ export async function createFrontendPlanLedgerTools(input) {
1359
1351
  entry: requirementSchema,
1360
1352
  }, { additionalProperties: false }),
1361
1353
  async execute(_toolCallId, params) {
1362
- await sleepRecordThrottle();
1363
1354
  const rawEntry = params?.entry;
1364
1355
  if (!isRecordObject(rawEntry)) {
1365
1356
  return planToolReceipt({
@@ -1387,11 +1378,26 @@ export async function createFrontendPlanLedgerTools(input) {
1387
1378
  void _omittedGap;
1388
1379
  entry = rest;
1389
1380
  }
1381
+ // Canonical-identity check: a requirement id outside the frozen
1382
+ // canonical list (e.g. a BR-* business rule picked up from the PRD
1383
+ // prose) would commit an immutable fact that finalize's
1384
+ // canonical-coverage gate rejects with no in-node cure. Reject here
1385
+ // and name the allowed ids.
1386
+ const id = typeof entry.id === "string" ? entry.id : "";
1387
+ if (id &&
1388
+ input.requirementIds &&
1389
+ input.requirementIds.length > 0 &&
1390
+ !input.requirementIds.includes(id)) {
1391
+ return planToolReceipt({
1392
+ ok: false,
1393
+ kind: "plan-requirement",
1394
+ error: `record_plan_requirement id "${id}" is not a frozen canonical requirement; canonical ids are: ${input.requirementIds.join(", ")}`,
1395
+ });
1396
+ }
1390
1397
  // A requirement id is a canonical identity: recording it twice would
1391
1398
  // compile a duplicate requirements[] entry and fail design review.
1392
1399
  // Reject duplicates at the tool boundary so the model can fix them
1393
1400
  // in-node instead of burning the attempt on a later validation error.
1394
- const id = typeof entry.id === "string" ? entry.id : "";
1395
1401
  if (id) {
1396
1402
  const existing = readCommittedEvents(store, attemptId).find((event) => {
1397
1403
  const fact = event.fact;
@@ -1422,7 +1428,6 @@ export async function createFrontendPlanLedgerTools(input) {
1422
1428
  entry: verificationTargetSchema,
1423
1429
  }, { additionalProperties: false }),
1424
1430
  async execute(_toolCallId, params) {
1425
- await sleepRecordThrottle();
1426
1431
  const rawEntry = params?.entry;
1427
1432
  if (!isRecordObject(rawEntry)) {
1428
1433
  return planToolReceipt({
@@ -1477,6 +1482,18 @@ export async function createFrontendPlanLedgerTools(input) {
1477
1482
  error: `record_plan_verification_target file is outside the writeSet patterns [${input.writeSetPatterns.join(", ")}]: ${rawEntry.file}; verification targets must point inside the task writeSet`,
1478
1483
  });
1479
1484
  }
1485
+ // Symbol shape check at the boundary: a fabricated symbol committed
1486
+ // here is immutable (duplicate ids are rejected), and while the
1487
+ // compile drops it, catching it now lets the model fix the target
1488
+ // in one receipt-free step.
1489
+ if (typeof rawEntry.symbol === "string" &&
1490
+ isSuspiciousVerificationSymbol(rawEntry.symbol)) {
1491
+ return planToolReceipt({
1492
+ ok: false,
1493
+ kind: "plan-verification-target",
1494
+ error: `record_plan_verification_target symbol "${rawEntry.symbol}" looks fabricated; use a real exported/describe/it symbol from ${rawEntry.file ?? "the target file"} or omit the symbol entirely (the trace gate verifies file+command)`,
1495
+ });
1496
+ }
1480
1497
  // uiStates: [] means this verification target is intentionally not
1481
1498
  // bound to a named UI state. Keep that canonical representation even
1482
1499
  // when a model omits the optional tool-boundary field.
@@ -1539,7 +1556,6 @@ export async function createFrontendPlanLedgerTools(input) {
1539
1556
  entry: evidenceGapSchema,
1540
1557
  }, { additionalProperties: false }),
1541
1558
  async execute(_toolCallId, params) {
1542
- await sleepRecordThrottle();
1543
1559
  const rawEntry = params?.entry;
1544
1560
  if (!isRecordObject(rawEntry)) {
1545
1561
  return planToolReceipt({
@@ -1732,6 +1748,19 @@ export async function createFrontendPlanLedgerTools(input) {
1732
1748
  const committed = readCommittedEvents(store, attemptId);
1733
1749
  await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"), committed);
1734
1750
  },
1751
+ committedFactCount: () => readCommittedEvents(store, attemptId).length,
1752
+ committedRequirementIds: () => {
1753
+ const ids = new Set();
1754
+ for (const event of readCommittedEvents(store, attemptId)) {
1755
+ const fact = event.fact;
1756
+ if (!fact || fact.kind !== "plan-requirement")
1757
+ continue;
1758
+ const entryFact = fact.entry;
1759
+ if (typeof entryFact?.id === "string")
1760
+ ids.add(entryFact.id);
1761
+ }
1762
+ return ids;
1763
+ },
1735
1764
  };
1736
1765
  }
1737
1766
  function isRecordObject(value) {
@@ -1869,7 +1898,6 @@ export async function createFrontendContractTools(input) {
1869
1898
  promptSnippet: `Commit 1-5 origin=contract ${kind} facts (up to 5 per message).`,
1870
1899
  parameters: Type.Object({}, { additionalProperties: true }),
1871
1900
  async execute(_toolCallId, params) {
1872
- await sleepRecordThrottle();
1873
1901
  const result = await adoptContractFact(kind, {
1874
1902
  kind,
1875
1903
  origin: "contract",
@@ -2524,6 +2552,171 @@ async function runFrontendDesignTerminalShadow(input) {
2524
2552
  }
2525
2553
  return input.mapped;
2526
2554
  }
2555
+ /** Tool subsets for the three frontend plan sessions (r17 split design). */
2556
+ const FRONTEND_PLAN_SEGMENTS = [
2557
+ {
2558
+ id: "coverage",
2559
+ toolNames: new Set([
2560
+ "record_plan_requirement",
2561
+ "record_plan_verification_target",
2562
+ "adopt_staged_fact",
2563
+ ]),
2564
+ instruction: [
2565
+ "PLAN SEGMENT 1/3 — coverage mapping only.",
2566
+ "Your ONLY job: for every frozen requirement, emit record_plan_requirement (requirement → implementation files) and record_plan_verification_target (verification target bound to requirement ids and files). Do NOT record components, UI states, mock, dependency, or routes — a follow-up session owns those.",
2567
+ "Do not call finalize_plan; it is not available in this segment.",
2568
+ ].join(" "),
2569
+ },
2570
+ {
2571
+ id: "ux-decisions",
2572
+ toolNames: new Set([
2573
+ "record_component_choice",
2574
+ "record_state_flow",
2575
+ "record_data_flow",
2576
+ "record_mock_api",
2577
+ "record_design_deviation",
2578
+ "record_dependency",
2579
+ "record_route_selection",
2580
+ "adopt_staged_fact",
2581
+ ]),
2582
+ instruction: [
2583
+ "PLAN SEGMENT 2/3 — UX decisions.",
2584
+ "Requirements and verification targets are already committed in the ledger (do NOT re-record them; duplicates are rejected). Your ONLY job: record component choices (decision=new requires sourceRequirementIds per the citation table), state flows, data flow, mock strategy, design deviation, dependency policy, and route selection.",
2585
+ "Do not call finalize_plan; it is not available in this segment.",
2586
+ ].join(" "),
2587
+ },
2588
+ {
2589
+ id: "finalize",
2590
+ toolNames: null,
2591
+ instruction: [
2592
+ "PLAN SEGMENT 3/3 — finalize.",
2593
+ "All record_* tools are available: if a finalize_plan receipt reports missing or invalid facts, fix them with the named record_* tool and call finalize_plan again. Otherwise call finalize_plan exactly once with no extra fields.",
2594
+ ].join(" "),
2595
+ },
2596
+ ];
2597
+ /**
2598
+ * Coverage-batch slicing for the frontend plan split (options 1+5): the
2599
+ * coverage segment becomes one session per requirement slice (default 4
2600
+ * requirements), so a small output budget can never be exhausted by
2601
+ * upfront reasoning about the whole requirement list. Zero-progress
2602
+ * batches are split in half and retried (option 5); single-requirement
2603
+ * zero-progress failures short-circuit to the retry ladder.
2604
+ */
2605
+ const FRONTEND_PLAN_COVERAGE_BATCH_SIZE = 4;
2606
+ const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 32;
2607
+ export async function runFrontendPlanSegmentedSessions(input) {
2608
+ const queue = [];
2609
+ // An empty list means "ledger unreadable / unknown" and must fall back to
2610
+ // one unscoped coverage session — only a non-empty list batches.
2611
+ const requirementIdsProvided = input.requirementIds !== undefined && input.requirementIds.length > 0;
2612
+ const pending = (input.requirementIds ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
2613
+ for (const segment of FRONTEND_PLAN_SEGMENTS) {
2614
+ if (segment.id !== "coverage") {
2615
+ queue.push({
2616
+ id: segment.id,
2617
+ toolNames: segment.toolNames,
2618
+ prompt: `${input.basePrompt}\n\n${segment.instruction}`,
2619
+ });
2620
+ continue;
2621
+ }
2622
+ // No requirement list (unreadable ledger) -> one unscoped coverage
2623
+ // session. A provided list with everything committed (resume) skips
2624
+ // coverage entirely.
2625
+ if (!requirementIdsProvided || pending.length > 0) {
2626
+ if (!requirementIdsProvided) {
2627
+ queue.push({
2628
+ id: segment.id,
2629
+ toolNames: segment.toolNames,
2630
+ prompt: `${input.basePrompt}\n\n${segment.instruction}`,
2631
+ });
2632
+ continue;
2633
+ }
2634
+ for (let i = 0; i < pending.length; i += FRONTEND_PLAN_COVERAGE_BATCH_SIZE) {
2635
+ const slice = pending.slice(i, i + FRONTEND_PLAN_COVERAGE_BATCH_SIZE);
2636
+ queue.push({
2637
+ id: `coverage-batch-${i / FRONTEND_PLAN_COVERAGE_BATCH_SIZE + 1}`,
2638
+ toolNames: segment.toolNames,
2639
+ coverageSlice: slice,
2640
+ prompt: `${input.basePrompt}\n\n${segment.instruction}\n\nCOVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them.`,
2641
+ });
2642
+ }
2643
+ }
2644
+ }
2645
+ let last;
2646
+ let index = 0;
2647
+ while (index < queue.length && index < FRONTEND_PLAN_BATCH_MAX_SESSIONS) {
2648
+ const session = queue[index];
2649
+ const remaining = (session.coverageSlice ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
2650
+ // Resume/earlier-batch commits may already cover this slice.
2651
+ if (session.coverageSlice && remaining.length === 0) {
2652
+ index += 1;
2653
+ continue;
2654
+ }
2655
+ let prompt = session.prompt;
2656
+ if (session.coverageSlice) {
2657
+ prompt = prompt.replace(/COVERAGE BATCH: process ONLY these requirements in this session: .*/, `COVERAGE BATCH: process ONLY these requirements in this session: ${remaining.join(", ")}. Other requirements are handled by separate sessions; do not record them.`);
2658
+ }
2659
+ const committedBefore = input.committedFactCount();
2660
+ const customTools = input.segmentCustomTools(session.toolNames);
2661
+ const result = await input.piStepFn({
2662
+ ...input.sessionOptions,
2663
+ prompt,
2664
+ ...(customTools.length > 0
2665
+ ? {
2666
+ writerToolPolicy: {
2667
+ requireSdk: true,
2668
+ customTools,
2669
+ },
2670
+ }
2671
+ : {}),
2672
+ });
2673
+ last = result;
2674
+ try {
2675
+ await input.flushLedger();
2676
+ }
2677
+ catch {
2678
+ // best-effort: the node-level flush runs again after the attempt
2679
+ }
2680
+ const committedAfter = input.committedFactCount();
2681
+ if (result.ok) {
2682
+ index += 1;
2683
+ continue;
2684
+ }
2685
+ const committedFactsOnlySuccess = session.id !== "finalize" &&
2686
+ !(result.assistantText ?? "").trim() &&
2687
+ !result.stderr.trim() &&
2688
+ !result.timedOut &&
2689
+ committedAfter > committedBefore;
2690
+ if (committedFactsOnlySuccess) {
2691
+ // The session died but banked facts: keep the progress and move on.
2692
+ index += 1;
2693
+ continue;
2694
+ }
2695
+ // Option 5: a multi-requirement coverage batch that failed with ZERO
2696
+ // new facts and no provider stderr is the upfront-reasoning burn —
2697
+ // halve the slice and retry instead of failing the attempt.
2698
+ const coverageSlice = session.coverageSlice;
2699
+ const zeroProgressBurn = coverageSlice !== undefined &&
2700
+ coverageSlice.length > 1 &&
2701
+ committedAfter === committedBefore &&
2702
+ !(result.assistantText ?? "").trim() &&
2703
+ !result.stderr.trim() &&
2704
+ !result.timedOut;
2705
+ if (zeroProgressBurn && coverageSlice) {
2706
+ const half = Math.ceil(coverageSlice.length / 2);
2707
+ queue.splice(index, 1, { ...session, coverageSlice: coverageSlice.slice(0, half) }, { ...session, coverageSlice: coverageSlice.slice(half) });
2708
+ continue;
2709
+ }
2710
+ return result;
2711
+ }
2712
+ return (last ?? {
2713
+ ok: false,
2714
+ stdout: "",
2715
+ stderr: "frontend plan segmentation produced no session",
2716
+ failureCategory: "empty-output",
2717
+ durationMs: 0,
2718
+ });
2719
+ }
2527
2720
  export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
2528
2721
  const started = Date.now();
2529
2722
  const persona = resolveDagPiPersona(input.task);
@@ -2906,10 +3099,9 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
2906
3099
  const piExtensionPaths = piExtensionsResolution && resolvePiBackend() !== "cli-only"
2907
3100
  ? piExtensionsResolution.resolved.flatMap((entry) => entry.entryPaths)
2908
3101
  : undefined;
2909
- result = await piStepFn({
3102
+ const piSessionOptions = {
2910
3103
  attachedFiles: [],
2911
3104
  modelConfig: resolveDagPiModelConfig(input.model, input.thinking ? { thinking: input.thinking } : undefined),
2912
- prompt: input.prompt,
2913
3105
  repoRoot: input.cwd,
2914
3106
  step,
2915
3107
  toolNames: resolveDagPiToolNames(input.task),
@@ -2922,7 +3114,6 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
2922
3114
  ...(input.task.contextBudget
2923
3115
  ? { contextBudget: input.task.contextBudget }
2924
3116
  : {}),
2925
- ...(writerToolPolicy ? { writerToolPolicy } : {}),
2926
3117
  ...(piExtensionPaths && piExtensionPaths.length > 0
2927
3118
  ? { piExtensionPaths }
2928
3119
  : {}),
@@ -2935,7 +3126,55 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
2935
3126
  bridgeActivity(activity.kind, activity.at);
2936
3127
  }
2937
3128
  : undefined,
2938
- });
3129
+ };
3130
+ if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
3131
+ // Frontend-only split: three sequential sessions with independent
3132
+ // output budgets (coverage -> UX decisions -> finalize), mirroring the
3133
+ // backend-test template's module sharding. Every other template keeps
3134
+ // the single-session path below.
3135
+ let planRequirementIds = [];
3136
+ try {
3137
+ const { readCommittedOriginFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
3138
+ const contractFacts = await readCommittedOriginFacts(meta.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl");
3139
+ // Only requirement facts: the contract ledger also carries
3140
+ // constraints (CON-*), evidence expectations (EV-*), handoff
3141
+ // intents (HND-*), open questions (OQ-*) and split proposals
3142
+ // (SPLIT-*) that all have ids — feeding those into the coverage
3143
+ // batches made the model record non-frozen plan-requirement ids
3144
+ // that finalize's canonical-coverage gate then rejected (r-ext2).
3145
+ planRequirementIds = contractFacts
3146
+ .filter((record) => record.fact
3147
+ ?.kind === "requirement")
3148
+ .map((record) => record.fact?.id)
3149
+ .filter((id) => typeof id === "string")
3150
+ .sort();
3151
+ }
3152
+ catch {
3153
+ // Unreadable ledger falls back to a single coverage session.
3154
+ }
3155
+ result = await runFrontendPlanSegmentedSessions({
3156
+ piStepFn,
3157
+ sessionOptions: piSessionOptions,
3158
+ basePrompt: input.prompt,
3159
+ attempt: input.attempt ?? 1,
3160
+ committedFactCount: () => planLedgerTools.committedFactCount(),
3161
+ requirementIds: planRequirementIds,
3162
+ committedRequirementIds: () => planLedgerTools.committedRequirementIds(),
3163
+ segmentCustomTools: (toolNames) => toolNames === null
3164
+ ? planLedgerTools.customTools
3165
+ : planLedgerTools.customTools.filter((tool) => typeof tool === "object" &&
3166
+ tool !== null &&
3167
+ toolNames.has(tool.name)),
3168
+ flushLedger: () => planLedgerTools.flush(),
3169
+ });
3170
+ }
3171
+ else {
3172
+ result = await piStepFn({
3173
+ ...piSessionOptions,
3174
+ prompt: input.prompt,
3175
+ ...(writerToolPolicy ? { writerToolPolicy } : {}),
3176
+ });
3177
+ }
2939
3178
  }
2940
3179
  catch (error) {
2941
3180
  if (playwrightToolContext) {
@@ -3072,25 +3311,10 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
3072
3311
  catch {
3073
3312
  // best-effort flush; missing ledger still fails at the node validator
3074
3313
  }
3075
- // Read-burst guard: a plan that burned dozens of read/grep/ls/find
3076
- // calls (re-reading upstream outputs and source files it should trust
3077
- // from typed facts) blows up the context window and eventually fails
3078
- // with 400 request-too-large. Detect it deterministically from the
3079
- // session log and retry with a reduced-reading instruction.
3080
- const readBurstIssues = input.task.readBudget?.onExhaustion === "return-guidance"
3081
- ? []
3082
- : await detectPlanReadBurst({
3083
- runDir: meta.runDir,
3084
- nodeId: input.task.id,
3085
- });
3086
- if (readBurstIssues.length > 0) {
3087
- return {
3088
- ...mapped,
3089
- ok: false,
3090
- failureCategory: "read-burst",
3091
- stderr: [mapped.stderr, ...readBurstIssues].filter(Boolean).join("\n\n"),
3092
- };
3093
- }
3314
+ // NOTE: the legacy plan read-burst guard lived here. It is dead code
3315
+ // since the plan node went tool-only (resolveDagPiToolNames grants no
3316
+ // read/grep/ls/find), so a read burst is structurally impossible; the
3317
+ // read-budget path (scout etc.) keeps its own guard.
3094
3318
  }
3095
3319
  if (!isWriteTask) {
3096
3320
  if (!mapped.ok &&
@@ -539,6 +539,18 @@ export const frontendImplementationContractSchema = z
539
539
  message: `unknown verification target ${targetId}`,
540
540
  path: ["requirements"],
541
541
  });
542
+ // Interactions bind to VTs too: an interaction whose behavior
543
+ // verification points at an unmaterialized VT is untraceable —
544
+ // r20 review finding (dangling *-BEHAVIOR references sailed
545
+ // through plan/design/implement to the final review).
546
+ for (const interaction of value.interactions)
547
+ for (const targetId of interaction.verificationTargetIds)
548
+ if (!verificationIds.includes(targetId))
549
+ ctx.addIssue({
550
+ code: "custom",
551
+ message: `interaction "${interaction.name}" references unknown verification target ${targetId}`,
552
+ path: ["interactions"],
553
+ });
542
554
  // Source fidelity ledger binding (AC-005/AC-006): 绑定携带
543
555
  // requirementToFragments(ledger v2)时,每个 requirement 必须携带
544
556
  // 非空 sourceFragmentIds,否则 fail closed(防引用伪造/缺失)。
@@ -570,12 +582,23 @@ export const frontendImplementationContractSchema = z
570
582
  if (state.applicable &&
571
583
  (!state.expectedBehavior ||
572
584
  !state.implementationTargets?.length ||
573
- !state.verificationTargetIds?.length))
585
+ !state.verificationTargetIds?.length)) {
586
+ // Name the state and the exact missing fields: the fixer is a
587
+ // model iterating on finalize receipts — it cannot fix a
588
+ // defect it cannot locate (r18: 39 blind finalize retries).
589
+ const missing = [];
590
+ if (!state.expectedBehavior)
591
+ missing.push("expectedBehavior");
592
+ if (!state.implementationTargets?.length)
593
+ missing.push("implementationTargets");
594
+ if (!state.verificationTargetIds?.length)
595
+ missing.push("verificationTargetIds");
574
596
  ctx.addIssue({
575
597
  code: "custom",
576
- message: "applicable UI state requires behavior, implementation, and verification",
598
+ message: `applicable UI state "${state.name}" is missing: ${missing.join(", ")} — record_state_flow it again with those fields filled`,
577
599
  path: ["uiStates"],
578
600
  });
601
+ }
579
602
  if (!state.applicable && !state.notApplicableReason)
580
603
  ctx.addIssue({
581
604
  code: "custom",
@@ -1826,12 +1849,25 @@ export async function analyzeFrontendImplementationContract(input) {
1826
1849
  if (parsedTargetFiles.some((file) => file.startsWith("/") || file.includes("\\")))
1827
1850
  fail("blocked", "invalid-output: frontend contract target paths must be relative POSIX paths", candidateJsonSha256);
1828
1851
  const parsedStates = asRecord(parsed)?.uiStates;
1829
- if (Array.isArray(parsedStates) && parsedStates.some((item) => {
1830
- const state = asRecord(item);
1831
- return state?.applicable === true &&
1832
- (!asString(state.expectedBehavior) || asStringArray(state.implementationTargets).length === 0 || asStringArray(state.verificationTargetIds).length === 0);
1833
- }))
1834
- fail("retryable-invalid", "invalid-output: applicable UI state requires behavior, implementation, and verification", candidateJsonSha256);
1852
+ if (Array.isArray(parsedStates)) {
1853
+ const incompleteStates = [];
1854
+ for (const item of parsedStates) {
1855
+ const state = asRecord(item);
1856
+ if (!state || state.applicable !== true)
1857
+ continue;
1858
+ const missing = [];
1859
+ if (!asString(state.expectedBehavior))
1860
+ missing.push("expectedBehavior");
1861
+ if (asStringArray(state.implementationTargets).length === 0)
1862
+ missing.push("implementationTargets");
1863
+ if (asStringArray(state.verificationTargetIds).length === 0)
1864
+ missing.push("verificationTargetIds");
1865
+ if (missing.length > 0)
1866
+ incompleteStates.push(`"${asString(state.name)}": missing ${missing.join(", ")}`);
1867
+ }
1868
+ if (incompleteStates.length > 0)
1869
+ fail("retryable-invalid", `invalid-output: applicable UI states incomplete — re-record each with record_state_flow filling the named fields: ${incompleteStates.join("; ")}`, candidateJsonSha256);
1870
+ }
1835
1871
  const parsedMockApi = asRecord(parsed)?.mockApi;
1836
1872
  if (asRecord(parsedMockApi) &&
1837
1873
  typeof asRecord(parsedMockApi)?.strategy === "string" &&
@@ -2079,7 +2115,7 @@ const VERIFICATION_SYMBOL_MAX_CHARS = 60;
2079
2115
  * loading") and `describe(...)` / `it(...)` forms stay valid — a real symbol
2080
2116
  * may be a function name, a dotted path, or a describe/it title.
2081
2117
  */
2082
- function isSuspiciousVerificationSymbol(symbol) {
2118
+ export function isSuspiciousVerificationSymbol(symbol) {
2083
2119
  const trimmed = symbol.trim();
2084
2120
  if (!trimmed)
2085
2121
  return false;
@@ -2126,6 +2162,17 @@ export class PlanPolicyPrecheckFailure extends Error {
2126
2162
  */
2127
2163
  export async function analyzeFrontendPlanPatchCandidate(input) {
2128
2164
  const analysis = await analyzeFrontendImplementationContract(input);
2165
+ // A committed VT with a fabricated symbol is an immutable ledger fact —
2166
+ // the record boundary rejects duplicate ids, so the model cannot overwrite
2167
+ // it and throwing here deadlocks the receipt loop (r19: VT-AC006-BEHAVIOR).
2168
+ // Drop suspicious symbols deterministically instead: the VT stays valid and
2169
+ // the deterministic trace gate verifies file+command (and resolvability)
2170
+ // after verification. Mirrors the shell materialization's drop semantics.
2171
+ for (const target of analysis.canonical.verificationTargets) {
2172
+ if (target.symbol && isSuspiciousVerificationSymbol(target.symbol)) {
2173
+ target.symbol = undefined;
2174
+ }
2175
+ }
2129
2176
  // Front-load the verification-symbol shape check so fabricated symbols
2130
2177
  // are fixed by the plan retry ladder in-node instead of failing the
2131
2178
  // verify trace gate at the end of the run.
@@ -287,9 +287,20 @@ export async function runFrontendReviewContextGate(input) {
287
287
  if (!parsedContract.success) {
288
288
  throw new FrontendReviewContextFailure("review-context-invalid-contract", "frontend review context invalid implementation contract");
289
289
  }
290
+ // Authorized diff surface = deliverable files + every verification-target
291
+ // file the contract references. The writer legitimately writes VT test
292
+ // artifacts (behavior verification), and the reviewer must audit their
293
+ // diffs; scoping this to targets.files alone failed the baseline-overlap
294
+ // check on app.test.js (r19 extreme smoke).
295
+ const authorizedChangedPaths = [
296
+ ...new Set([
297
+ ...parsedContract.data.targets.files,
298
+ ...parsedContract.data.verificationTargets.map((target) => target.file),
299
+ ]),
300
+ ];
290
301
  const diff = await runFrontendWorktreeDiffGate({
291
302
  ...input,
292
- authorizedChangedPaths: parsedContract.data.targets.files,
303
+ authorizedChangedPaths,
293
304
  });
294
305
  const verificationTrace = await readRequiredJson(input.runDir, "contracts/frontend-verification-trace.json");
295
306
  const repairAssessment = await readOptionalRepairAssessment(input.runDir);
@@ -260,7 +260,12 @@ export function restorePlanPatchFromCommittedFacts(records) {
260
260
  if (isRecord(fact.patch))
261
261
  return fact.patch;
262
262
  }
263
- return undefined;
263
+ // No finalize-published snapshot: the receipt fix loop can end an attempt
264
+ // before any finalize_plan succeeds while dozens of record_* facts are
265
+ // already committed. Assemble the patch from those record facts instead of
266
+ // declaring the ledger missing (r19: "ledger missing" discarded a ledger
267
+ // with 92 committed facts).
268
+ return assemblePlanPatchFromCommittedFacts(records);
264
269
  }
265
270
  /**
266
271
  * A+B (AC-004): reverse of `planLedgerFactsFromPatch` for the incremental
@@ -280,12 +285,33 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
280
285
  PLAN_LEDGER_FACT_KINDS.includes(fact.kind));
281
286
  if (facts.length === 0)
282
287
  return undefined;
283
- for (const fact of facts) {
284
- if (isRecord(fact.patch))
285
- return fact.patch;
286
- }
288
+ // Singleton record facts: the LAST committed fact wins. The model corrects
289
+ // a rejected plan by re-recording the fact (r18: 39 finalize retries never
290
+ // converged because the first state-flow fact kept shadowing the
291
+ // corrections). Requirements / verification targets / evidence gaps stay
292
+ // additive (aggregated below); component choices accumulate; singletons
293
+ // supersede.
294
+ const lastByKind = (kind) => {
295
+ let found;
296
+ for (const fact of facts) {
297
+ if (fact.kind === kind)
298
+ found = fact;
299
+ }
300
+ return found;
301
+ };
302
+ // Patch snapshots (published by earlier finalize calls) carry no routes;
303
+ // exclude them so a snapshot cannot shadow the model's route selection.
304
+ const routeSurface = (() => {
305
+ let found;
306
+ for (const fact of facts) {
307
+ if (fact.kind !== "target-surface" || isRecord(fact.patch))
308
+ continue;
309
+ found = fact;
310
+ }
311
+ return found;
312
+ })();
287
313
  const patch = {};
288
- const targetSurface = facts.find((fact) => fact.kind === "target-surface");
314
+ const targetSurface = routeSurface;
289
315
  if (targetSurface) {
290
316
  const routes = canonicalStringSet(asStringArray(targetSurface.routes));
291
317
  if (routes.length > 0)
@@ -295,14 +321,34 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
295
321
  const collectedChoices = componentChoices.flatMap((fact) => Array.isArray(fact.uiComponentChoices)
296
322
  ? fact.uiComponentChoices.filter(isRecord)
297
323
  : []);
324
+ // Later corrections supersede earlier submissions per purpose: without
325
+ // this, a re-recorded choice (fixing a wrong specReference) would appear
326
+ // twice and the reviewer would still audit the stale entry (r19).
327
+ const choicesByPurpose = new Map();
328
+ const dedupedChoices = [];
329
+ for (const choice of collectedChoices) {
330
+ const purpose = asString(choice.purpose);
331
+ if (!purpose) {
332
+ dedupedChoices.push(choice);
333
+ continue;
334
+ }
335
+ if (choicesByPurpose.has(purpose)) {
336
+ const at = dedupedChoices.findIndex((existing) => asString(existing.purpose) === purpose);
337
+ dedupedChoices[at] = choice;
338
+ }
339
+ else {
340
+ choicesByPurpose.set(purpose, choice);
341
+ dedupedChoices.push(choice);
342
+ }
343
+ }
298
344
  const stylingStrategy = componentChoices
299
345
  .map((fact) => asString(fact.stylingStrategy))
300
346
  .find((value) => value.length > 0);
301
- if (collectedChoices.length > 0)
302
- patch.uiComponentChoices = collectedChoices;
347
+ if (dedupedChoices.length > 0)
348
+ patch.uiComponentChoices = dedupedChoices;
303
349
  if (stylingStrategy)
304
350
  patch.stylingStrategy = stylingStrategy;
305
- const stateFlow = facts.find((fact) => fact.kind === "state-flow");
351
+ const stateFlow = lastByKind("state-flow");
306
352
  if (stateFlow) {
307
353
  if (Array.isArray(stateFlow.uiStates)) {
308
354
  patch.uiStates = stateFlow.uiStates.filter(isRecord);
@@ -311,17 +357,17 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
311
357
  patch.interactions = stateFlow.interactions.filter(isRecord);
312
358
  }
313
359
  }
314
- const mockApi = facts.find((fact) => fact.kind === "mock-api");
360
+ const mockApi = lastByKind("mock-api");
315
361
  if (mockApi && isRecord(mockApi.mockApi)) {
316
362
  patch.mockApi = mockApi.mockApi;
317
363
  }
318
- const deviation = facts.find((fact) => fact.kind === "design-deviation");
364
+ const deviation = lastByKind("design-deviation");
319
365
  if (deviation) {
320
366
  const conflicts = canonicalStringSet(asStringArray(deviation.conflicts));
321
367
  if (conflicts.length > 0)
322
368
  patch.designEvidence = { conflicts };
323
369
  }
324
- const dependency = facts.find((fact) => fact.kind === "dependency");
370
+ const dependency = lastByKind("dependency");
325
371
  const dependencyPolicy = asString(dependency?.policy);
326
372
  if (dependencyPolicy)
327
373
  patch.dependencyPolicy = dependencyPolicy;
@@ -3210,6 +3210,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3210
3210
  ...(requiresOpenspecClassification ? ["When a component choice uses an OpenSpec selection, cite that selection; otherwise do not classify unrelated candidates."] : []),
3211
3211
  "Call finalize_plan exactly once after the necessary typed facts. Return no Markdown narrative.",
3212
3212
  "TOOL-ONLY PLAN: Do not read Contract/Scout stdout, task sources, or Scout-confirmed target files. Contract and Scout already own evidence discovery; use the injected upstream facts, record a genuine evidence gap when those facts are insufficient, and start committing record_* facts immediately. For decision=new, pass sourceRequirementIds to record_component_choice; runtime derives the exact PRD citation from the frozen ledger.",
3213
+ "Output budget protocol (hard, max output <=16K per turn): never enumerate-reason the whole requirement list before your first record_* call — that reasoning burns the entire output budget and the attempt dies with zero committed facts. Process requirements in order: think about ONE requirement briefly, immediately emit its record calls (up to 5 per message), then move to the next. If your budget runs low, stop recording and call finalize_plan with what is committed — the retry ladder continues the remainder in a fresh session.",
3213
3214
  fixedVerificationContext,
3214
3215
  scopedOpenspecContext,
3215
3216
  mockContextBlock,
@@ -315,9 +315,11 @@ export const dagFrontendVerificationBundleSchema = z
315
315
  behaviorEvidence: dagShellVerifyEvidenceSchema,
316
316
  lintBaselineNodeId: dagFrontendNodeIdSchema.optional(),
317
317
  writerNodeIds: z.array(dagFrontendNodeIdSchema).optional(),
318
- /** M8: verify-shell only ever runs in initial mode; a failure is terminal
319
- * (routes to recovery) and never selects a same-run repair branch. */
320
- mode: z.literal("initial"),
318
+ /** "initial" = first-pass assess bundle; "repair" = the convergence
319
+ * loop's post-repair reverify, which re-runs the same frozen commands
320
+ * and fails on any failure (still no same-run repair branch inside the
321
+ * node itself). */
322
+ mode: z.enum(["initial", "repair"]),
321
323
  })
322
324
  .superRefine((bundle, context) => {
323
325
  const groups = [
package/harness.json CHANGED
@@ -83,8 +83,8 @@
83
83
  "pi": {
84
84
  "description": "Pi 负责规划、评审、诊断;当 DAG toolProfile=write 时也可做有界写入。模型按复杂度三档配置,支持 provider/model 字符串或带 thinking 的对象;斜杠前为 Pi provider,后为 modelId,勿只写裸 modelId。",
85
85
  "LOW": "wizard-local/minimax-m3",
86
- "MED": "wizard-local/gpt-5.6-sol",
87
- "HIGH": "wizard-local/gpt-5.6-sol"
86
+ "MED": "wizard-local/gpt-5.5",
87
+ "HIGH": "wizard-local/gpt-5.5"
88
88
  }
89
89
  }
90
- }
90
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tea-agent/loop-agent",
3
- "version": "0.39.0-beta.13",
3
+ "version": "0.39.0-beta.14",
4
4
  "type": "module",
5
5
  "bin": {
6
6
  "loop-agent": "bin/loop-agent.js",