@tea-agent/loop-agent 0.39.0-beta.13 → 0.39.0-beta.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +1 -0
- package/dist/application/task-lifecycle/advance.js +20 -4
- package/dist/application/task-lifecycle/observe.js +171 -17
- package/dist/application/task-lifecycle/plan-transitions.js +42 -7
- package/dist/build-stamp.json +3 -3
- package/dist/executors/dag-pi-executor.js +308 -84
- package/dist/workflows/dag/frontend-implementation-contract.js +56 -9
- package/dist/workflows/dag/frontend-review-context.js +12 -1
- package/dist/workflows/dag/frontend-shadow-dual-write.js +58 -12
- package/dist/workflows/dag/init-hybrid.js +1 -0
- package/dist/workflows/dag/types.js +5 -3
- package/harness.json +3 -3
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# 更新日志
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
|
+
- 0.39.0-beta.14(frontend-node dogfood):plan-pi 三段分段会话 + coverage 需求批次切片(4/批、零进度减半)、finalize 收据修复环(聚合诊断+修复 JSON)、record_* 边界校验左移(VT 重复/writeSet/交互 VT 引用/规范需求 id/可疑 symbol)、编译装配 last-wins(组件按 purpose 去重修正)、review-context 授权面扩展至 VT 测试文件、设计评审修订降级为 admission advisory。
|
|
4
5
|
- 前端 DAG 极端环境加固(16KB 输出 / 120KB 上下文真实冒烟,r2-r5 验证):`frontend-contract-pi` 改为消费编译后的 ledger 输入块(`<frontend_contract_input>`,12KB 上限、cap ladder、requirement id 永不丢)并移除 read 工具,以增量 `record_*` 提交为唯一输出模式(OUTPUT BUDGET DISCIPLINE),修复极端小输出窗口下模型反复 read 源文件空转 1h54m 超时的问题(r3 起 contract 3m54s 正常完成、37 次 record_* / 0 次 read);plan/contract 的 `record_*` 工具支持每 message 最多 5 条批量提交并加 1s 调用间隔,rate-limit 退避上限提高到 60s,缓解密集小调用触发第三方网关限流;修复 fragment-inventory 对 V4 标题式 AC(`#### AC-NNN`)的双重 emit 缺陷(同一标题行被 section 入口与正文 heading 分支各切一次,30 个 AC 结构性产生 30 个 duplicate)。
|
|
5
6
|
|
|
6
7
|
- 修复连续 `dag rerun-task` 丢失前代评审意见的问题:直接父 run 未走到 review/design review 时,`deriveDagRerunFeedback` 沿 run-owned lineage 有界追溯最近 5 层并继承最近的 committed typed findings;较近的 `approve_review` / `approve_design` 作为对应 resolution barrier,防止已解决 finding 被复活。lineage run ID、生命周期目录与 `state.json.runId` 必须一致,损坏、越界、循环或超深链路 fail closed。
|
|
@@ -260,6 +260,10 @@ async function monitorRun(input) {
|
|
|
260
260
|
failureCategory: "monitor-timeout",
|
|
261
261
|
});
|
|
262
262
|
}
|
|
263
|
+
function isStartupGateEligible(snapshot) {
|
|
264
|
+
const status = snapshot.source?.sourceFidelityStartup?.status;
|
|
265
|
+
return status === undefined || status === "complete" || status === "not-applicable";
|
|
266
|
+
}
|
|
263
267
|
function nextMutationAction(actions) {
|
|
264
268
|
for (const action of actions) {
|
|
265
269
|
if (action.type !== "stop")
|
|
@@ -379,7 +383,10 @@ export async function advanceTaskLifecycle(input) {
|
|
|
379
383
|
};
|
|
380
384
|
}
|
|
381
385
|
const token = parseGateToken(input.rejectGateToken);
|
|
382
|
-
const currentGate = snapshot.gate ??
|
|
386
|
+
const currentGate = snapshot.gate ??
|
|
387
|
+
(isStartupGateEligible(snapshot)
|
|
388
|
+
? await buildCurrentGate(input.repoRoot, input.taskId)
|
|
389
|
+
: null);
|
|
383
390
|
if (!currentGate || !gateMatches(currentGate, token)) {
|
|
384
391
|
return {
|
|
385
392
|
taskId: input.taskId,
|
|
@@ -443,11 +450,12 @@ export async function advanceTaskLifecycle(input) {
|
|
|
443
450
|
dryRun,
|
|
444
451
|
skipFinalize: Boolean(input.skipFinalize),
|
|
445
452
|
forcePrepareContract: Boolean(input.fromDraftPath) && !preparedContractThisCall,
|
|
453
|
+
startupRepairAttempted: preparedContractThisCall,
|
|
446
454
|
});
|
|
447
455
|
if (dryRun) {
|
|
448
456
|
const plan = planOnce(snapshot);
|
|
449
457
|
const plannedGate = snapshot.gate ??
|
|
450
|
-
(snapshot.dagDraft?.exists
|
|
458
|
+
(isStartupGateEligible(snapshot) && snapshot.dagDraft?.exists
|
|
451
459
|
? await buildCurrentGate(input.repoRoot, input.taskId)
|
|
452
460
|
: null);
|
|
453
461
|
return {
|
|
@@ -586,6 +594,10 @@ export async function advanceTaskLifecycle(input) {
|
|
|
586
594
|
continue;
|
|
587
595
|
}
|
|
588
596
|
if (action.type === "prepare-contract") {
|
|
597
|
+
const startupStatus = snapshot.source?.sourceFidelityStartup?.status;
|
|
598
|
+
const startupRepairAction = startupStatus === "missing" ||
|
|
599
|
+
startupStatus === "invalid" ||
|
|
600
|
+
startupStatus === "coverage-incomplete";
|
|
589
601
|
// 生产 source fidelity Pi 接线(AC-PROD-001/005):仅 frontend-implementation
|
|
590
602
|
// 任务注入 reconcile executor + 独立 reviewer;ineligible 返回 undefined 不接线。
|
|
591
603
|
// 主路径与 semantic-intake retry 路径共用同一接线,保证任意一条生产路径都实际
|
|
@@ -731,7 +743,9 @@ export async function advanceTaskLifecycle(input) {
|
|
|
731
743
|
message: "structured requirement draft via semantic intake; projected managed source",
|
|
732
744
|
});
|
|
733
745
|
preparedContractThisCall = true;
|
|
734
|
-
if (
|
|
746
|
+
if (startupRepairAction ||
|
|
747
|
+
refreshManagedContractThisCall ||
|
|
748
|
+
input.fromDraftPath) {
|
|
735
749
|
refreshManagedContractThisCall = false;
|
|
736
750
|
regenerateDagThisCall = true;
|
|
737
751
|
}
|
|
@@ -825,7 +839,9 @@ export async function advanceTaskLifecycle(input) {
|
|
|
825
839
|
break;
|
|
826
840
|
}
|
|
827
841
|
preparedContractThisCall = true;
|
|
828
|
-
if (
|
|
842
|
+
if (startupRepairAction ||
|
|
843
|
+
refreshManagedContractThisCall ||
|
|
844
|
+
input.fromDraftPath) {
|
|
829
845
|
// The old DAG and gate were bound to the pre-mutation contract.
|
|
830
846
|
refreshManagedContractThisCall = false;
|
|
831
847
|
regenerateDagThisCall = true;
|
|
@@ -1,9 +1,11 @@
|
|
|
1
|
-
import { access, readFile } from "node:fs/promises";
|
|
1
|
+
import { access, readFile, readdir } from "node:fs/promises";
|
|
2
2
|
import path from "node:path";
|
|
3
3
|
import { getTaskDir, getTaskPaths, getTaskStatus, loadTaskConfig, } from "../../task/runtime.js";
|
|
4
4
|
import { logicalRevisionForState, observeTaskContractState, } from "../../task/contract/index.js";
|
|
5
5
|
import { assessDagRunLiveness, listAllDagRunEntries, readDagRunSpec, readDagRunState, } from "../../workflows/dag/lifecycle.js";
|
|
6
6
|
import { resolveRunningControllerIdentity } from "../../shared/package-metadata.js";
|
|
7
|
+
import { parseLedgerJson, validateRequirementLedger, } from "../../task/source-prepare/ledger.js";
|
|
8
|
+
import { REQUIREMENT_LEDGER_FILE_NAME } from "../../task/source-prepare/types.js";
|
|
7
9
|
import { loadLifecycleRecord } from "./record.js";
|
|
8
10
|
import { buildGateBinding, buildWriteSetGate, buildWriteSetGateDelta, controllerIdentityAnchor, hashCanonicalPayload, normalizeWriteSet, sha256Hex, } from "./gates.js";
|
|
9
11
|
async function exists(filePath) {
|
|
@@ -47,6 +49,140 @@ async function taskExists(repoRoot, taskId) {
|
|
|
47
49
|
const paths = getTaskPaths(repoRoot, taskId);
|
|
48
50
|
return exists(paths.taskConfigPath);
|
|
49
51
|
}
|
|
52
|
+
async function assessReferenceEvidence(referencesDir) {
|
|
53
|
+
try {
|
|
54
|
+
const entries = await readdir(referencesDir);
|
|
55
|
+
const importedPrdCount = entries.filter((name) => name.endsWith(".md")).length;
|
|
56
|
+
return importedPrdCount > 0
|
|
57
|
+
? { status: "present", importedPrdCount }
|
|
58
|
+
: { status: "absent-or-empty", importedPrdCount: 0 };
|
|
59
|
+
}
|
|
60
|
+
catch (error) {
|
|
61
|
+
if (error &&
|
|
62
|
+
typeof error === "object" &&
|
|
63
|
+
"code" in error &&
|
|
64
|
+
error.code === "ENOENT") {
|
|
65
|
+
return { status: "absent-or-empty", importedPrdCount: 0 };
|
|
66
|
+
}
|
|
67
|
+
return {
|
|
68
|
+
status: "unknown",
|
|
69
|
+
importedPrdCount: 0,
|
|
70
|
+
error: error instanceof Error ? error.message : String(error),
|
|
71
|
+
};
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
function hasLedgerStructure(value) {
|
|
75
|
+
if (!value || typeof value !== "object")
|
|
76
|
+
return false;
|
|
77
|
+
const ledger = value;
|
|
78
|
+
return (typeof ledger.taskId === "string" &&
|
|
79
|
+
typeof ledger.inputDigest === "string" &&
|
|
80
|
+
typeof ledger.requirementSha256 === "string" &&
|
|
81
|
+
Array.isArray(ledger.sourcePaths) &&
|
|
82
|
+
Array.isArray(ledger.fragments) &&
|
|
83
|
+
Array.isArray(ledger.canonicalRequirements) &&
|
|
84
|
+
Boolean(ledger.stats && typeof ledger.stats === "object") &&
|
|
85
|
+
Array.isArray(ledger.unresolved) &&
|
|
86
|
+
Array.isArray(ledger.conflicts) &&
|
|
87
|
+
Array.isArray(ledger.sourceResolutions) &&
|
|
88
|
+
typeof ledger.schemaReconcilerVersion === "string" &&
|
|
89
|
+
typeof ledger.builtAt === "string");
|
|
90
|
+
}
|
|
91
|
+
async function assessSourceFidelityStartup(input) {
|
|
92
|
+
if (!input.applicable) {
|
|
93
|
+
return {
|
|
94
|
+
status: "not-applicable",
|
|
95
|
+
ledgerPath: input.ledgerPath,
|
|
96
|
+
diagnostics: [],
|
|
97
|
+
};
|
|
98
|
+
}
|
|
99
|
+
if (input.referenceEvidenceError) {
|
|
100
|
+
return {
|
|
101
|
+
status: "invalid",
|
|
102
|
+
ledgerPath: input.ledgerPath,
|
|
103
|
+
diagnostics: [
|
|
104
|
+
{
|
|
105
|
+
code: "SOURCE_FIDELITY_LEDGER_STALE",
|
|
106
|
+
message: `imported PRD reference evidence is unreadable: ${input.referenceEvidenceError}`,
|
|
107
|
+
},
|
|
108
|
+
],
|
|
109
|
+
};
|
|
110
|
+
}
|
|
111
|
+
let raw;
|
|
112
|
+
try {
|
|
113
|
+
raw = await readFile(input.ledgerPath, "utf-8");
|
|
114
|
+
}
|
|
115
|
+
catch (error) {
|
|
116
|
+
const missing = !(await exists(input.ledgerPath));
|
|
117
|
+
return {
|
|
118
|
+
status: missing ? "missing" : "invalid",
|
|
119
|
+
ledgerPath: input.ledgerPath,
|
|
120
|
+
diagnostics: [
|
|
121
|
+
{
|
|
122
|
+
code: missing
|
|
123
|
+
? "SOURCE_FIDELITY_LEDGER_MISSING"
|
|
124
|
+
: "SOURCE_FIDELITY_LEDGER_STALE",
|
|
125
|
+
message: missing
|
|
126
|
+
? "managed frontend task has no requirement-ledger.json"
|
|
127
|
+
: `requirement ledger is unreadable: ${error instanceof Error ? error.message : String(error)}`,
|
|
128
|
+
},
|
|
129
|
+
],
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
try {
|
|
133
|
+
const ledger = parseLedgerJson(raw);
|
|
134
|
+
if (!hasLedgerStructure(ledger)) {
|
|
135
|
+
return {
|
|
136
|
+
status: "invalid",
|
|
137
|
+
ledgerPath: input.ledgerPath,
|
|
138
|
+
diagnostics: [
|
|
139
|
+
{
|
|
140
|
+
code: "SOURCE_FIDELITY_LEDGER_STALE",
|
|
141
|
+
message: "requirement ledger is missing required structural fields",
|
|
142
|
+
},
|
|
143
|
+
],
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
const nonDecisionErrors = validateRequirementLedger(ledger, {
|
|
147
|
+
allowedUnresolved: true,
|
|
148
|
+
allowedConflicts: true,
|
|
149
|
+
});
|
|
150
|
+
if (nonDecisionErrors.length > 0) {
|
|
151
|
+
const coverageCodes = new Set([
|
|
152
|
+
"SOURCE_FIDELITY_HIGH_RISK_UNCOVERED",
|
|
153
|
+
"SOURCE_FIDELITY_REQUIREMENT_NOT_COVERED",
|
|
154
|
+
]);
|
|
155
|
+
return {
|
|
156
|
+
status: nonDecisionErrors.every((error) => coverageCodes.has(error.code))
|
|
157
|
+
? "coverage-incomplete"
|
|
158
|
+
: "invalid",
|
|
159
|
+
ledgerPath: input.ledgerPath,
|
|
160
|
+
diagnostics: nonDecisionErrors,
|
|
161
|
+
};
|
|
162
|
+
}
|
|
163
|
+
const decisionErrors = validateRequirementLedger(ledger);
|
|
164
|
+
if (decisionErrors.length > 0) {
|
|
165
|
+
return {
|
|
166
|
+
status: "human-decision-blocked",
|
|
167
|
+
ledgerPath: input.ledgerPath,
|
|
168
|
+
diagnostics: decisionErrors,
|
|
169
|
+
};
|
|
170
|
+
}
|
|
171
|
+
return { status: "complete", ledgerPath: input.ledgerPath, diagnostics: [] };
|
|
172
|
+
}
|
|
173
|
+
catch (error) {
|
|
174
|
+
return {
|
|
175
|
+
status: "invalid",
|
|
176
|
+
ledgerPath: input.ledgerPath,
|
|
177
|
+
diagnostics: [
|
|
178
|
+
{
|
|
179
|
+
code: "SOURCE_FIDELITY_LEDGER_STALE",
|
|
180
|
+
message: error instanceof Error ? error.message : String(error),
|
|
181
|
+
},
|
|
182
|
+
],
|
|
183
|
+
};
|
|
184
|
+
}
|
|
185
|
+
}
|
|
50
186
|
/**
|
|
51
187
|
* Explicit Task↔DAG association only:
|
|
52
188
|
* - lifecycle.json recorded runAssociations
|
|
@@ -182,10 +318,12 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
|
|
|
182
318
|
const record = await loadLifecycleRecord(repoRoot, taskId);
|
|
183
319
|
const contractState = await observeTaskContractState(repoRoot, taskId);
|
|
184
320
|
let title;
|
|
321
|
+
let taskKind;
|
|
185
322
|
let taskConfigAllowed = [];
|
|
186
323
|
try {
|
|
187
324
|
const config = await loadTaskConfig(repoRoot, taskId);
|
|
188
325
|
title = config.title;
|
|
326
|
+
taskKind = config.taskKind;
|
|
189
327
|
taskConfigAllowed = config.allowedPaths ?? [];
|
|
190
328
|
}
|
|
191
329
|
catch {
|
|
@@ -199,19 +337,22 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
|
|
|
199
337
|
...(requirementReady ? [] : [requirementPath]),
|
|
200
338
|
...(constraintsReady ? [] : [constraintsPath]),
|
|
201
339
|
];
|
|
202
|
-
|
|
340
|
+
const contractManaged = contractState.effectiveStatus === "managed" || Boolean(contractState.ref);
|
|
203
341
|
const referencesDir = path.join(paths.sourceDir, "references");
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
342
|
+
const referenceEvidence = await assessReferenceEvidence(referencesDir);
|
|
343
|
+
const importedPrdCount = referenceEvidence.importedPrdCount;
|
|
344
|
+
const ledgerPath = path.join(paths.sourceDir, REQUIREMENT_LEDGER_FILE_NAME);
|
|
345
|
+
const sourceFidelityStartup = await assessSourceFidelityStartup({
|
|
346
|
+
ledgerPath,
|
|
347
|
+
applicable: contractManaged &&
|
|
348
|
+
taskKind === "frontend-implementation" &&
|
|
349
|
+
referenceEvidence.status !== "absent-or-empty",
|
|
350
|
+
referenceEvidenceError: referenceEvidence.status === "unknown"
|
|
351
|
+
? referenceEvidence.error
|
|
352
|
+
: undefined,
|
|
353
|
+
});
|
|
354
|
+
const sourceFidelityReady = sourceFidelityStartup.status === "complete" ||
|
|
355
|
+
sourceFidelityStartup.status === "not-applicable";
|
|
215
356
|
const dagDraftPath = paths.dagDraftPath;
|
|
216
357
|
const dagDraftExists = await exists(dagDraftPath);
|
|
217
358
|
let specHash = null;
|
|
@@ -285,7 +426,11 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
|
|
|
285
426
|
// Open gate only when draft is strict-validated and the current binding is
|
|
286
427
|
// not already approved/rejected. Rejected same-digest gates stay closed
|
|
287
428
|
// until contract/source/DAG/writeSet/controller binding changes.
|
|
288
|
-
const gate = gateBound &&
|
|
429
|
+
const gate = gateBound &&
|
|
430
|
+
strictValidated &&
|
|
431
|
+
sourceFidelityReady &&
|
|
432
|
+
!approvedCurrent &&
|
|
433
|
+
!rejectedCurrent
|
|
289
434
|
? buildWriteSetGate({ taskId, bound: gateBound, delta: gateDelta })
|
|
290
435
|
: null;
|
|
291
436
|
const gateOpen = Boolean(gate);
|
|
@@ -310,11 +455,10 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
|
|
|
310
455
|
activeRun.failureCategory === "needs-attention" ||
|
|
311
456
|
activeRun.failureCategory === "suspected-stall" ||
|
|
312
457
|
activeRun.failureCategory === "monitor-timeout"));
|
|
313
|
-
const contractManaged = contractState.effectiveStatus === "managed" || Boolean(contractState.ref);
|
|
314
458
|
const lifecycleState = deriveLifecycleState({
|
|
315
459
|
exists: true,
|
|
316
460
|
contractManaged,
|
|
317
|
-
sourceReady: requirementReady,
|
|
461
|
+
sourceReady: requirementReady && sourceFidelityReady,
|
|
318
462
|
dagValidated: strictValidated,
|
|
319
463
|
gateOpen: Boolean(gate),
|
|
320
464
|
activeRun,
|
|
@@ -352,6 +496,7 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
|
|
|
352
496
|
constraintsReady,
|
|
353
497
|
importedPrdCount,
|
|
354
498
|
missing,
|
|
499
|
+
sourceFidelityStartup,
|
|
355
500
|
},
|
|
356
501
|
dagDraft: {
|
|
357
502
|
path: dagDraftPath,
|
|
@@ -375,7 +520,16 @@ export async function observeTaskLifecycle(repoRoot, taskId) {
|
|
|
375
520
|
exists: closeoutExists || Boolean(record?.closeout),
|
|
376
521
|
},
|
|
377
522
|
completedTransitions: record?.completedTransitions ?? [],
|
|
378
|
-
blockers:
|
|
523
|
+
blockers: sourceFidelityReady
|
|
524
|
+
? []
|
|
525
|
+
: sourceFidelityStartup.diagnostics.map((diagnostic) => ({
|
|
526
|
+
code: diagnostic.code,
|
|
527
|
+
message: diagnostic.message,
|
|
528
|
+
details: {
|
|
529
|
+
startupStatus: sourceFidelityStartup.status,
|
|
530
|
+
ledgerPath: sourceFidelityStartup.ledgerPath,
|
|
531
|
+
},
|
|
532
|
+
})),
|
|
379
533
|
warnings: [],
|
|
380
534
|
next,
|
|
381
535
|
};
|
|
@@ -46,6 +46,37 @@ export function planTransitions(input) {
|
|
|
46
46
|
actions.push({ type: "recover-contract" });
|
|
47
47
|
expectedTransitions.push("contract-recovered");
|
|
48
48
|
}
|
|
49
|
+
const startupStatus = snapshot.source?.sourceFidelityStartup?.status;
|
|
50
|
+
const startupRepairDetected = startupStatus === "missing" ||
|
|
51
|
+
startupStatus === "invalid" ||
|
|
52
|
+
startupStatus === "coverage-incomplete";
|
|
53
|
+
if (startupRepairDetected && input.startupRepairAttempted) {
|
|
54
|
+
actions.push({ type: "stop", reason: "blocker" });
|
|
55
|
+
return {
|
|
56
|
+
actions,
|
|
57
|
+
gate: null,
|
|
58
|
+
blockers,
|
|
59
|
+
next: {
|
|
60
|
+
kind: "advance",
|
|
61
|
+
command: `loop-agent task advance ${snapshot.taskId} --json`,
|
|
62
|
+
},
|
|
63
|
+
expectedTransitions,
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
const startupRepairRequired = startupRepairDetected;
|
|
67
|
+
if (startupStatus === "human-decision-blocked") {
|
|
68
|
+
actions.push({ type: "stop", reason: "blocker" });
|
|
69
|
+
return {
|
|
70
|
+
actions,
|
|
71
|
+
gate: null,
|
|
72
|
+
blockers,
|
|
73
|
+
next: {
|
|
74
|
+
kind: "external-action",
|
|
75
|
+
description: "Requirement ledger requires an existing human source decision; preserve sourceResolutions, resolve the blocker, then advance again.",
|
|
76
|
+
},
|
|
77
|
+
expectedTransitions,
|
|
78
|
+
};
|
|
79
|
+
}
|
|
49
80
|
const sourceReady = snapshot.source?.requirementReady === true;
|
|
50
81
|
const contractManaged = snapshot.contract?.effectiveStatus === "managed" ||
|
|
51
82
|
Boolean(snapshot.contract?.revision && snapshot.contract.revision > 0);
|
|
@@ -53,7 +84,9 @@ export function planTransitions(input) {
|
|
|
53
84
|
// Prefer prepare once PRDs are on disk; re-import only when caller still
|
|
54
85
|
// signals hasPrdInput (advance.ts latches this after one import per call).
|
|
55
86
|
const shouldImportPrd = input.hasPrdInput;
|
|
56
|
-
if (input.forcePrepareContract ||
|
|
87
|
+
if (input.forcePrepareContract ||
|
|
88
|
+
input.refreshManagedContract ||
|
|
89
|
+
startupRepairRequired) {
|
|
57
90
|
if (shouldImportPrd) {
|
|
58
91
|
actions.push({ type: "import-prd" });
|
|
59
92
|
expectedTransitions.push("prd-imported");
|
|
@@ -124,7 +157,8 @@ export function planTransitions(input) {
|
|
|
124
157
|
const strictValidated = snapshot.dagDraft?.strictValidated === true;
|
|
125
158
|
const dagRefreshRequired = Boolean(input.forceDagRefresh ||
|
|
126
159
|
input.refreshManagedContract ||
|
|
127
|
-
input.forcePrepareContract
|
|
160
|
+
input.forcePrepareContract ||
|
|
161
|
+
startupRepairRequired);
|
|
128
162
|
if (!dagExists || dagRefreshRequired) {
|
|
129
163
|
actions.push({ type: "generate-dag" });
|
|
130
164
|
expectedTransitions.push("dag-generated");
|
|
@@ -145,10 +179,11 @@ export function planTransitions(input) {
|
|
|
145
179
|
snapshot.lifecycleState === "evidence-closed" ||
|
|
146
180
|
snapshot.lifecycleState === "needs-attention" ||
|
|
147
181
|
snapshot.lifecycleState === "awaiting-decision";
|
|
148
|
-
const
|
|
182
|
+
const approveCurrentGate = input.approveGate && !dagRefreshRequired;
|
|
183
|
+
const approvedForCurrent = postRunState || (approveCurrentGate && gate !== null);
|
|
149
184
|
if (!approvedForCurrent && (strictValidated || !dagExists || dagRefreshRequired)) {
|
|
150
185
|
// Gate opens after generate+validate in the advance loop; plan includes open.
|
|
151
|
-
if (!
|
|
186
|
+
if (!approveCurrentGate) {
|
|
152
187
|
actions.push({ type: "open-write-set-gate" });
|
|
153
188
|
expectedTransitions.push("write-set-gate-opened");
|
|
154
189
|
if (input.dryRun) {
|
|
@@ -177,14 +212,14 @@ export function planTransitions(input) {
|
|
|
177
212
|
(snapshot.lifecycleState === "run-failed" && !failedRecoverable) ||
|
|
178
213
|
snapshot.lifecycleState === "evidence-closed";
|
|
179
214
|
if (!alreadyTerminalRun) {
|
|
180
|
-
if (
|
|
215
|
+
if (approveCurrentGate && !postRunState) {
|
|
181
216
|
expectedTransitions.push("write-set-approved");
|
|
182
217
|
actions.push({ type: "start-or-monitor-run" });
|
|
183
218
|
expectedTransitions.push("dag-run-started", "dag-run-monitored");
|
|
184
219
|
}
|
|
185
220
|
else if (snapshot.lifecycleState === "running" ||
|
|
186
221
|
snapshot.activeRun ||
|
|
187
|
-
(
|
|
222
|
+
(approveCurrentGate && Boolean(snapshot.activeRun))) {
|
|
188
223
|
actions.push({ type: "start-or-monitor-run" });
|
|
189
224
|
expectedTransitions.push("dag-run-monitored");
|
|
190
225
|
}
|
|
@@ -218,7 +253,7 @@ export function planTransitions(input) {
|
|
|
218
253
|
actions,
|
|
219
254
|
gate,
|
|
220
255
|
blockers,
|
|
221
|
-
next: deriveNext(snapshot, gate,
|
|
256
|
+
next: deriveNext(snapshot, gate, approveCurrentGate),
|
|
222
257
|
expectedTransitions,
|
|
223
258
|
};
|
|
224
259
|
}
|
package/dist/build-stamp.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"schemaVersion": 1,
|
|
3
|
-
"version": "0.39.0-beta.
|
|
4
|
-
"gitSha": "
|
|
5
|
-
"builtAt": "2026-08-
|
|
3
|
+
"version": "0.39.0-beta.14",
|
|
4
|
+
"gitSha": "9d8bebca76a2608d9b38a1af1f379fd5fe5b04b0",
|
|
5
|
+
"builtAt": "2026-08-30T09:13:51.707Z"
|
|
6
6
|
}
|
|
@@ -16,6 +16,7 @@ import { redactPromptForLog, truncateOutput, } from "../shared/output-truncation
|
|
|
16
16
|
import { GitStatusUnavailableError, pathsChangedDuringRun, readGitStatusPorcelain, recoverRootNulArtifact, snapshotGitStatusPathFingerprints, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
|
|
17
17
|
import { captureWorkspaceWriteSnapshot, diffWorkspaceWriteSnapshots, } from "./workspace-write-snapshot.js";
|
|
18
18
|
import { pathMatchesPattern } from "../shared/git-progress.js";
|
|
19
|
+
import { isSuspiciousVerificationSymbol } from "../workflows/dag/frontend-implementation-contract.js";
|
|
19
20
|
import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
|
|
20
21
|
import { assessBackendTestMdPlanCompleteness, assessBackendTestMdWriterCompleteness, assessBackendTestPytestPlanCompleteness, assessBackendTestPytestWriterCompleteness, assessBackendTestShardChildCompleteness, backendTestWriterProgressRoleForTask, classifyBackendTestWriterCompletenessFailure, isBackendTestCompletenessRetryCandidate, isBackendTestMdPlanTask, isBackendTestPytestPlanTask, isBackendTestShardChildTask, writeBackendTestWriterProgressArtifacts, } from "../workflows/dag/backend-test-writer-completeness.js";
|
|
21
22
|
import { resolveBackendTestLayout } from "../workflows/dag/backend-test-layout.js";
|
|
@@ -236,16 +237,6 @@ export const FRONTEND_PLAN_RECORD_TOOL_NAMES = [
|
|
|
236
237
|
];
|
|
237
238
|
export const FRONTEND_PLAN_TERMINAL_TOOL_NAMES = ["finalize_plan"];
|
|
238
239
|
export const FRONTEND_PLAN_ADOPT_TOOL_NAMES = ["adopt_staged_fact"];
|
|
239
|
-
/**
|
|
240
|
-
* Inter-call delay for high-frequency frontend record_* tools. Incremental
|
|
241
|
-
* submission (one model response per 1-5 entries) means dozens of sequential
|
|
242
|
-
* API round trips in a few minutes; third-party gateways (huu.dqy.ink) trip
|
|
243
|
-
* rate limits whose windows exceed normal API RPM even at ~13 calls/min. A
|
|
244
|
-
* short fixed delay before each tool body keeps the sustained rate under
|
|
245
|
-
* typical thresholds while batching cuts the total call count.
|
|
246
|
-
*/
|
|
247
|
-
const FRONTEND_RECORD_TOOL_THROTTLE_MS = 1000;
|
|
248
|
-
const sleepRecordThrottle = () => new Promise((resolve) => setTimeout(resolve, FRONTEND_RECORD_TOOL_THROTTLE_MS));
|
|
249
240
|
/** M5: `frontend-review-pi` emits its authoritative terminal verdict through
|
|
250
241
|
* committed typed tools instead of the legacy JSON verdict parse. */
|
|
251
242
|
export function isFrontendReviewTypedTerminalNode(task) {
|
|
@@ -395,10 +386,6 @@ export function scanReviewTerminalKindsFromSessionEvents(content) {
|
|
|
395
386
|
}
|
|
396
387
|
return kinds;
|
|
397
388
|
}
|
|
398
|
-
/** Read-only discovery tool budget for frontend-plan-pi. Exceeding it means
|
|
399
|
-
* the plan re-read upstream outputs/sources instead of trusting typed facts,
|
|
400
|
-
* which blows up the context window (400 request-too-large). */
|
|
401
|
-
const PLAN_READ_TOOL_BUDGET = 40;
|
|
402
389
|
const READ_ONLY_TOOL_NAMES = new Set(["read", "grep", "ls", "find"]);
|
|
403
390
|
function readEventPath(event) {
|
|
404
391
|
const candidates = [event.path, event.readPath, event.input];
|
|
@@ -483,45 +470,6 @@ export async function detectNodeReadBudget(input) {
|
|
|
483
470
|
issues.push(`frontend ${input.nodeId} read budget exceeded: ${stats.elapsedMs}ms (budget ${input.budget.maxMs}ms)`);
|
|
484
471
|
return issues;
|
|
485
472
|
}
|
|
486
|
-
/** Deterministic read-burst guard for frontend-plan-pi: count read-only
|
|
487
|
-
* discovery tool calls (read/grep/ls/find) from the session log. Over budget →
|
|
488
|
-
* read-burst, retried with a reduced-reading instruction. Pure scan; a
|
|
489
|
-
* successful plan under budget is never blocked. */
|
|
490
|
-
export async function detectPlanReadBurst(input) {
|
|
491
|
-
const sessionEventsPath = path.join(input.runDir, input.nodeId, "session-events.jsonl");
|
|
492
|
-
let count = 0;
|
|
493
|
-
try {
|
|
494
|
-
const content = await readFile(sessionEventsPath, "utf8");
|
|
495
|
-
for (const line of content.split("\n")) {
|
|
496
|
-
if (!line.trim())
|
|
497
|
-
continue;
|
|
498
|
-
try {
|
|
499
|
-
const event = JSON.parse(line);
|
|
500
|
-
if (event.type === "tool_execution_start" &&
|
|
501
|
-
typeof event.toolName === "string" &&
|
|
502
|
-
(event.toolName === "read" ||
|
|
503
|
-
event.toolName === "grep" ||
|
|
504
|
-
event.toolName === "ls" ||
|
|
505
|
-
event.toolName === "find")) {
|
|
506
|
-
count += 1;
|
|
507
|
-
}
|
|
508
|
-
}
|
|
509
|
-
catch {
|
|
510
|
-
// skip unparseable line
|
|
511
|
-
}
|
|
512
|
-
}
|
|
513
|
-
}
|
|
514
|
-
catch {
|
|
515
|
-
// Missing/unreadable session log → no burst detection
|
|
516
|
-
return [];
|
|
517
|
-
}
|
|
518
|
-
if (count > PLAN_READ_TOOL_BUDGET) {
|
|
519
|
-
return [
|
|
520
|
-
`frontend plan read-burst: ${count} read-only tool calls (budget ${PLAN_READ_TOOL_BUDGET}). Trust the upstream contract/scout typed facts; do not re-read contract/scout outputs or source files already captured. Minimize discovery reads, commit record_* facts directly, then finalize_plan.`,
|
|
521
|
-
];
|
|
522
|
-
}
|
|
523
|
-
return [];
|
|
524
|
-
}
|
|
525
473
|
export const FRONTEND_DESIGN_TERMINAL_TOOL_NAMES = new Set([
|
|
526
474
|
"approve_design",
|
|
527
475
|
"request_design_changes",
|
|
@@ -932,11 +880,27 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
932
880
|
import("typebox"),
|
|
933
881
|
import("@earendil-works/pi-coding-agent"),
|
|
934
882
|
]);
|
|
935
|
-
const { readCommittedEvents, writeTypedEventStoreJsonl } = await import("../workflows/dag/frontend-typed-event-store.js");
|
|
883
|
+
const { loadTypedEventStore, readCommittedEvents, writeTypedEventStoreJsonl, } = await import("../workflows/dag/frontend-typed-event-store.js");
|
|
936
884
|
const { adoptStagedFact, adoptTypedEventFact, stageTypedEventFact } = await import("../workflows/dag/frontend-typed-event-transaction.js");
|
|
937
885
|
const { assemblePlanPatchFromCommittedFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
|
|
938
886
|
const store = input.store;
|
|
939
887
|
const attemptId = input.attemptId;
|
|
888
|
+
// A retry creates a fresh executor-local store, but the plan ledger is the
|
|
889
|
+
// cross-attempt authority. Restore the committed prefix before registering
|
|
890
|
+
// tools; otherwise the first flush of a retry can overwrite facts that the
|
|
891
|
+
// previous attempt had already committed. The on-disk file contains only
|
|
892
|
+
// committed records, so loading it is also fail-closed with respect to
|
|
893
|
+
// staged/quarantined facts.
|
|
894
|
+
const persisted = await loadTypedEventStore(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"));
|
|
895
|
+
if (persisted.records.length > 0) {
|
|
896
|
+
const existingEventIds = new Set(store.records.map((record) => record.eventId));
|
|
897
|
+
for (const record of persisted.records) {
|
|
898
|
+
if (!existingEventIds.has(record.eventId)) {
|
|
899
|
+
store.records.push(record);
|
|
900
|
+
}
|
|
901
|
+
}
|
|
902
|
+
store.revision = Math.max(store.revision, persisted.revision);
|
|
903
|
+
}
|
|
940
904
|
const stringArray = Type.Array(Type.String({}));
|
|
941
905
|
const optionalString = Type.Optional(Type.String({}));
|
|
942
906
|
const optionalStringArray = Type.Optional(stringArray);
|
|
@@ -1127,7 +1091,6 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1127
1091
|
stylingStrategy: optionalString,
|
|
1128
1092
|
}, { additionalProperties: false }),
|
|
1129
1093
|
async execute(_toolCallId, params) {
|
|
1130
|
-
await sleepRecordThrottle();
|
|
1131
1094
|
const rawChoice = params?.choice;
|
|
1132
1095
|
if (!isRecordObject(rawChoice)) {
|
|
1133
1096
|
return planToolReceipt({
|
|
@@ -1180,7 +1143,16 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1180
1143
|
? { stylingStrategy: params.stylingStrategy }
|
|
1181
1144
|
: {}),
|
|
1182
1145
|
});
|
|
1183
|
-
|
|
1146
|
+
// Echo the frozen citation the runtime derived: the model sees the
|
|
1147
|
+
// purpose↔citation mapping it just committed and can re-record the
|
|
1148
|
+
// choice (last-wins per purpose at compile) when it mismatches.
|
|
1149
|
+
const echo = {
|
|
1150
|
+
...result,
|
|
1151
|
+
...(choice.specReference
|
|
1152
|
+
? { derivedSpecReference: choice.specReference }
|
|
1153
|
+
: {}),
|
|
1154
|
+
};
|
|
1155
|
+
return planToolReceipt(echo);
|
|
1184
1156
|
},
|
|
1185
1157
|
});
|
|
1186
1158
|
const recordStateFlowTool = defineTool({
|
|
@@ -1264,6 +1236,26 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1264
1236
|
error: "record_state_flow requires non-empty interaction.name (id is accepted only as a legacy alias)",
|
|
1265
1237
|
});
|
|
1266
1238
|
}
|
|
1239
|
+
// Interaction -> VT forward references are legal only against
|
|
1240
|
+
// already-committed VT facts. In the segmented flow every VT
|
|
1241
|
+
// commits in the coverage segment before state flows run, so a
|
|
1242
|
+
// dangling reference here is a real defect (r20: *-BEHAVIOR
|
|
1243
|
+
// refs reached the final review untraceable).
|
|
1244
|
+
const interactionVtIds = stringList(interaction.verificationTargetIds);
|
|
1245
|
+
const committedVtIds = new Set(readCommittedEvents(store, attemptId)
|
|
1246
|
+
.map((event) => event.fact)
|
|
1247
|
+
.filter((fact) => fact.kind === "plan-verification-target")
|
|
1248
|
+
.map((fact) => fact.entry
|
|
1249
|
+
?.id)
|
|
1250
|
+
.filter((id) => typeof id === "string"));
|
|
1251
|
+
const unknownVtIds = interactionVtIds.filter((id) => !committedVtIds.has(id));
|
|
1252
|
+
if (unknownVtIds.length > 0) {
|
|
1253
|
+
return planToolReceipt({
|
|
1254
|
+
ok: false,
|
|
1255
|
+
kind: "state-flow",
|
|
1256
|
+
error: `record_state_flow interaction "${resolvedName}" references verification targets that are not recorded yet: ${unknownVtIds.join(", ")}; record them with record_plan_verification_target first, then re-record this state flow`,
|
|
1257
|
+
});
|
|
1258
|
+
}
|
|
1267
1259
|
interactions.push({ ...interaction, name: resolvedName });
|
|
1268
1260
|
}
|
|
1269
1261
|
const states = uiStates
|
|
@@ -1359,7 +1351,6 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1359
1351
|
entry: requirementSchema,
|
|
1360
1352
|
}, { additionalProperties: false }),
|
|
1361
1353
|
async execute(_toolCallId, params) {
|
|
1362
|
-
await sleepRecordThrottle();
|
|
1363
1354
|
const rawEntry = params?.entry;
|
|
1364
1355
|
if (!isRecordObject(rawEntry)) {
|
|
1365
1356
|
return planToolReceipt({
|
|
@@ -1387,11 +1378,26 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1387
1378
|
void _omittedGap;
|
|
1388
1379
|
entry = rest;
|
|
1389
1380
|
}
|
|
1381
|
+
// Canonical-identity check: a requirement id outside the frozen
|
|
1382
|
+
// canonical list (e.g. a BR-* business rule picked up from the PRD
|
|
1383
|
+
// prose) would commit an immutable fact that finalize's
|
|
1384
|
+
// canonical-coverage gate rejects with no in-node cure. Reject here
|
|
1385
|
+
// and name the allowed ids.
|
|
1386
|
+
const id = typeof entry.id === "string" ? entry.id : "";
|
|
1387
|
+
if (id &&
|
|
1388
|
+
input.requirementIds &&
|
|
1389
|
+
input.requirementIds.length > 0 &&
|
|
1390
|
+
!input.requirementIds.includes(id)) {
|
|
1391
|
+
return planToolReceipt({
|
|
1392
|
+
ok: false,
|
|
1393
|
+
kind: "plan-requirement",
|
|
1394
|
+
error: `record_plan_requirement id "${id}" is not a frozen canonical requirement; canonical ids are: ${input.requirementIds.join(", ")}`,
|
|
1395
|
+
});
|
|
1396
|
+
}
|
|
1390
1397
|
// A requirement id is a canonical identity: recording it twice would
|
|
1391
1398
|
// compile a duplicate requirements[] entry and fail design review.
|
|
1392
1399
|
// Reject duplicates at the tool boundary so the model can fix them
|
|
1393
1400
|
// in-node instead of burning the attempt on a later validation error.
|
|
1394
|
-
const id = typeof entry.id === "string" ? entry.id : "";
|
|
1395
1401
|
if (id) {
|
|
1396
1402
|
const existing = readCommittedEvents(store, attemptId).find((event) => {
|
|
1397
1403
|
const fact = event.fact;
|
|
@@ -1422,7 +1428,6 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1422
1428
|
entry: verificationTargetSchema,
|
|
1423
1429
|
}, { additionalProperties: false }),
|
|
1424
1430
|
async execute(_toolCallId, params) {
|
|
1425
|
-
await sleepRecordThrottle();
|
|
1426
1431
|
const rawEntry = params?.entry;
|
|
1427
1432
|
if (!isRecordObject(rawEntry)) {
|
|
1428
1433
|
return planToolReceipt({
|
|
@@ -1477,6 +1482,18 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1477
1482
|
error: `record_plan_verification_target file is outside the writeSet patterns [${input.writeSetPatterns.join(", ")}]: ${rawEntry.file}; verification targets must point inside the task writeSet`,
|
|
1478
1483
|
});
|
|
1479
1484
|
}
|
|
1485
|
+
// Symbol shape check at the boundary: a fabricated symbol committed
|
|
1486
|
+
// here is immutable (duplicate ids are rejected), and while the
|
|
1487
|
+
// compile drops it, catching it now lets the model fix the target
|
|
1488
|
+
// in one receipt-free step.
|
|
1489
|
+
if (typeof rawEntry.symbol === "string" &&
|
|
1490
|
+
isSuspiciousVerificationSymbol(rawEntry.symbol)) {
|
|
1491
|
+
return planToolReceipt({
|
|
1492
|
+
ok: false,
|
|
1493
|
+
kind: "plan-verification-target",
|
|
1494
|
+
error: `record_plan_verification_target symbol "${rawEntry.symbol}" looks fabricated; use a real exported/describe/it symbol from ${rawEntry.file ?? "the target file"} or omit the symbol entirely (the trace gate verifies file+command)`,
|
|
1495
|
+
});
|
|
1496
|
+
}
|
|
1480
1497
|
// uiStates: [] means this verification target is intentionally not
|
|
1481
1498
|
// bound to a named UI state. Keep that canonical representation even
|
|
1482
1499
|
// when a model omits the optional tool-boundary field.
|
|
@@ -1539,7 +1556,6 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1539
1556
|
entry: evidenceGapSchema,
|
|
1540
1557
|
}, { additionalProperties: false }),
|
|
1541
1558
|
async execute(_toolCallId, params) {
|
|
1542
|
-
await sleepRecordThrottle();
|
|
1543
1559
|
const rawEntry = params?.entry;
|
|
1544
1560
|
if (!isRecordObject(rawEntry)) {
|
|
1545
1561
|
return planToolReceipt({
|
|
@@ -1732,6 +1748,19 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1732
1748
|
const committed = readCommittedEvents(store, attemptId);
|
|
1733
1749
|
await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"), committed);
|
|
1734
1750
|
},
|
|
1751
|
+
committedFactCount: () => readCommittedEvents(store, attemptId).length,
|
|
1752
|
+
committedRequirementIds: () => {
|
|
1753
|
+
const ids = new Set();
|
|
1754
|
+
for (const event of readCommittedEvents(store, attemptId)) {
|
|
1755
|
+
const fact = event.fact;
|
|
1756
|
+
if (!fact || fact.kind !== "plan-requirement")
|
|
1757
|
+
continue;
|
|
1758
|
+
const entryFact = fact.entry;
|
|
1759
|
+
if (typeof entryFact?.id === "string")
|
|
1760
|
+
ids.add(entryFact.id);
|
|
1761
|
+
}
|
|
1762
|
+
return ids;
|
|
1763
|
+
},
|
|
1735
1764
|
};
|
|
1736
1765
|
}
|
|
1737
1766
|
function isRecordObject(value) {
|
|
@@ -1869,7 +1898,6 @@ export async function createFrontendContractTools(input) {
|
|
|
1869
1898
|
promptSnippet: `Commit 1-5 origin=contract ${kind} facts (up to 5 per message).`,
|
|
1870
1899
|
parameters: Type.Object({}, { additionalProperties: true }),
|
|
1871
1900
|
async execute(_toolCallId, params) {
|
|
1872
|
-
await sleepRecordThrottle();
|
|
1873
1901
|
const result = await adoptContractFact(kind, {
|
|
1874
1902
|
kind,
|
|
1875
1903
|
origin: "contract",
|
|
@@ -2524,6 +2552,171 @@ async function runFrontendDesignTerminalShadow(input) {
|
|
|
2524
2552
|
}
|
|
2525
2553
|
return input.mapped;
|
|
2526
2554
|
}
|
|
2555
|
+
/** Tool subsets for the three frontend plan sessions (r17 split design). */
|
|
2556
|
+
const FRONTEND_PLAN_SEGMENTS = [
|
|
2557
|
+
{
|
|
2558
|
+
id: "coverage",
|
|
2559
|
+
toolNames: new Set([
|
|
2560
|
+
"record_plan_requirement",
|
|
2561
|
+
"record_plan_verification_target",
|
|
2562
|
+
"adopt_staged_fact",
|
|
2563
|
+
]),
|
|
2564
|
+
instruction: [
|
|
2565
|
+
"PLAN SEGMENT 1/3 — coverage mapping only.",
|
|
2566
|
+
"Your ONLY job: for every frozen requirement, emit record_plan_requirement (requirement → implementation files) and record_plan_verification_target (verification target bound to requirement ids and files). Do NOT record components, UI states, mock, dependency, or routes — a follow-up session owns those.",
|
|
2567
|
+
"Do not call finalize_plan; it is not available in this segment.",
|
|
2568
|
+
].join(" "),
|
|
2569
|
+
},
|
|
2570
|
+
{
|
|
2571
|
+
id: "ux-decisions",
|
|
2572
|
+
toolNames: new Set([
|
|
2573
|
+
"record_component_choice",
|
|
2574
|
+
"record_state_flow",
|
|
2575
|
+
"record_data_flow",
|
|
2576
|
+
"record_mock_api",
|
|
2577
|
+
"record_design_deviation",
|
|
2578
|
+
"record_dependency",
|
|
2579
|
+
"record_route_selection",
|
|
2580
|
+
"adopt_staged_fact",
|
|
2581
|
+
]),
|
|
2582
|
+
instruction: [
|
|
2583
|
+
"PLAN SEGMENT 2/3 — UX decisions.",
|
|
2584
|
+
"Requirements and verification targets are already committed in the ledger (do NOT re-record them; duplicates are rejected). Your ONLY job: record component choices (decision=new requires sourceRequirementIds per the citation table), state flows, data flow, mock strategy, design deviation, dependency policy, and route selection.",
|
|
2585
|
+
"Do not call finalize_plan; it is not available in this segment.",
|
|
2586
|
+
].join(" "),
|
|
2587
|
+
},
|
|
2588
|
+
{
|
|
2589
|
+
id: "finalize",
|
|
2590
|
+
toolNames: null,
|
|
2591
|
+
instruction: [
|
|
2592
|
+
"PLAN SEGMENT 3/3 — finalize.",
|
|
2593
|
+
"All record_* tools are available: if a finalize_plan receipt reports missing or invalid facts, fix them with the named record_* tool and call finalize_plan again. Otherwise call finalize_plan exactly once with no extra fields.",
|
|
2594
|
+
].join(" "),
|
|
2595
|
+
},
|
|
2596
|
+
];
|
|
2597
|
+
/**
|
|
2598
|
+
* Coverage-batch slicing for the frontend plan split (options 1+5): the
|
|
2599
|
+
* coverage segment becomes one session per requirement slice (default 4
|
|
2600
|
+
* requirements), so a small output budget can never be exhausted by
|
|
2601
|
+
* upfront reasoning about the whole requirement list. Zero-progress
|
|
2602
|
+
* batches are split in half and retried (option 5); single-requirement
|
|
2603
|
+
* zero-progress failures short-circuit to the retry ladder.
|
|
2604
|
+
*/
|
|
2605
|
+
const FRONTEND_PLAN_COVERAGE_BATCH_SIZE = 4;
|
|
2606
|
+
const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 32;
|
|
2607
|
+
export async function runFrontendPlanSegmentedSessions(input) {
|
|
2608
|
+
const queue = [];
|
|
2609
|
+
// An empty list means "ledger unreadable / unknown" and must fall back to
|
|
2610
|
+
// one unscoped coverage session — only a non-empty list batches.
|
|
2611
|
+
const requirementIdsProvided = input.requirementIds !== undefined && input.requirementIds.length > 0;
|
|
2612
|
+
const pending = (input.requirementIds ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
|
|
2613
|
+
for (const segment of FRONTEND_PLAN_SEGMENTS) {
|
|
2614
|
+
if (segment.id !== "coverage") {
|
|
2615
|
+
queue.push({
|
|
2616
|
+
id: segment.id,
|
|
2617
|
+
toolNames: segment.toolNames,
|
|
2618
|
+
prompt: `${input.basePrompt}\n\n${segment.instruction}`,
|
|
2619
|
+
});
|
|
2620
|
+
continue;
|
|
2621
|
+
}
|
|
2622
|
+
// No requirement list (unreadable ledger) -> one unscoped coverage
|
|
2623
|
+
// session. A provided list with everything committed (resume) skips
|
|
2624
|
+
// coverage entirely.
|
|
2625
|
+
if (!requirementIdsProvided || pending.length > 0) {
|
|
2626
|
+
if (!requirementIdsProvided) {
|
|
2627
|
+
queue.push({
|
|
2628
|
+
id: segment.id,
|
|
2629
|
+
toolNames: segment.toolNames,
|
|
2630
|
+
prompt: `${input.basePrompt}\n\n${segment.instruction}`,
|
|
2631
|
+
});
|
|
2632
|
+
continue;
|
|
2633
|
+
}
|
|
2634
|
+
for (let i = 0; i < pending.length; i += FRONTEND_PLAN_COVERAGE_BATCH_SIZE) {
|
|
2635
|
+
const slice = pending.slice(i, i + FRONTEND_PLAN_COVERAGE_BATCH_SIZE);
|
|
2636
|
+
queue.push({
|
|
2637
|
+
id: `coverage-batch-${i / FRONTEND_PLAN_COVERAGE_BATCH_SIZE + 1}`,
|
|
2638
|
+
toolNames: segment.toolNames,
|
|
2639
|
+
coverageSlice: slice,
|
|
2640
|
+
prompt: `${input.basePrompt}\n\n${segment.instruction}\n\nCOVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them.`,
|
|
2641
|
+
});
|
|
2642
|
+
}
|
|
2643
|
+
}
|
|
2644
|
+
}
|
|
2645
|
+
let last;
|
|
2646
|
+
let index = 0;
|
|
2647
|
+
while (index < queue.length && index < FRONTEND_PLAN_BATCH_MAX_SESSIONS) {
|
|
2648
|
+
const session = queue[index];
|
|
2649
|
+
const remaining = (session.coverageSlice ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
|
|
2650
|
+
// Resume/earlier-batch commits may already cover this slice.
|
|
2651
|
+
if (session.coverageSlice && remaining.length === 0) {
|
|
2652
|
+
index += 1;
|
|
2653
|
+
continue;
|
|
2654
|
+
}
|
|
2655
|
+
let prompt = session.prompt;
|
|
2656
|
+
if (session.coverageSlice) {
|
|
2657
|
+
prompt = prompt.replace(/COVERAGE BATCH: process ONLY these requirements in this session: .*/, `COVERAGE BATCH: process ONLY these requirements in this session: ${remaining.join(", ")}. Other requirements are handled by separate sessions; do not record them.`);
|
|
2658
|
+
}
|
|
2659
|
+
const committedBefore = input.committedFactCount();
|
|
2660
|
+
const customTools = input.segmentCustomTools(session.toolNames);
|
|
2661
|
+
const result = await input.piStepFn({
|
|
2662
|
+
...input.sessionOptions,
|
|
2663
|
+
prompt,
|
|
2664
|
+
...(customTools.length > 0
|
|
2665
|
+
? {
|
|
2666
|
+
writerToolPolicy: {
|
|
2667
|
+
requireSdk: true,
|
|
2668
|
+
customTools,
|
|
2669
|
+
},
|
|
2670
|
+
}
|
|
2671
|
+
: {}),
|
|
2672
|
+
});
|
|
2673
|
+
last = result;
|
|
2674
|
+
try {
|
|
2675
|
+
await input.flushLedger();
|
|
2676
|
+
}
|
|
2677
|
+
catch {
|
|
2678
|
+
// best-effort: the node-level flush runs again after the attempt
|
|
2679
|
+
}
|
|
2680
|
+
const committedAfter = input.committedFactCount();
|
|
2681
|
+
if (result.ok) {
|
|
2682
|
+
index += 1;
|
|
2683
|
+
continue;
|
|
2684
|
+
}
|
|
2685
|
+
const committedFactsOnlySuccess = session.id !== "finalize" &&
|
|
2686
|
+
!(result.assistantText ?? "").trim() &&
|
|
2687
|
+
!result.stderr.trim() &&
|
|
2688
|
+
!result.timedOut &&
|
|
2689
|
+
committedAfter > committedBefore;
|
|
2690
|
+
if (committedFactsOnlySuccess) {
|
|
2691
|
+
// The session died but banked facts: keep the progress and move on.
|
|
2692
|
+
index += 1;
|
|
2693
|
+
continue;
|
|
2694
|
+
}
|
|
2695
|
+
// Option 5: a multi-requirement coverage batch that failed with ZERO
|
|
2696
|
+
// new facts and no provider stderr is the upfront-reasoning burn —
|
|
2697
|
+
// halve the slice and retry instead of failing the attempt.
|
|
2698
|
+
const coverageSlice = session.coverageSlice;
|
|
2699
|
+
const zeroProgressBurn = coverageSlice !== undefined &&
|
|
2700
|
+
coverageSlice.length > 1 &&
|
|
2701
|
+
committedAfter === committedBefore &&
|
|
2702
|
+
!(result.assistantText ?? "").trim() &&
|
|
2703
|
+
!result.stderr.trim() &&
|
|
2704
|
+
!result.timedOut;
|
|
2705
|
+
if (zeroProgressBurn && coverageSlice) {
|
|
2706
|
+
const half = Math.ceil(coverageSlice.length / 2);
|
|
2707
|
+
queue.splice(index, 1, { ...session, coverageSlice: coverageSlice.slice(0, half) }, { ...session, coverageSlice: coverageSlice.slice(half) });
|
|
2708
|
+
continue;
|
|
2709
|
+
}
|
|
2710
|
+
return result;
|
|
2711
|
+
}
|
|
2712
|
+
return (last ?? {
|
|
2713
|
+
ok: false,
|
|
2714
|
+
stdout: "",
|
|
2715
|
+
stderr: "frontend plan segmentation produced no session",
|
|
2716
|
+
failureCategory: "empty-output",
|
|
2717
|
+
durationMs: 0,
|
|
2718
|
+
});
|
|
2719
|
+
}
|
|
2527
2720
|
export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
|
|
2528
2721
|
const started = Date.now();
|
|
2529
2722
|
const persona = resolveDagPiPersona(input.task);
|
|
@@ -2906,10 +3099,9 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
2906
3099
|
const piExtensionPaths = piExtensionsResolution && resolvePiBackend() !== "cli-only"
|
|
2907
3100
|
? piExtensionsResolution.resolved.flatMap((entry) => entry.entryPaths)
|
|
2908
3101
|
: undefined;
|
|
2909
|
-
|
|
3102
|
+
const piSessionOptions = {
|
|
2910
3103
|
attachedFiles: [],
|
|
2911
3104
|
modelConfig: resolveDagPiModelConfig(input.model, input.thinking ? { thinking: input.thinking } : undefined),
|
|
2912
|
-
prompt: input.prompt,
|
|
2913
3105
|
repoRoot: input.cwd,
|
|
2914
3106
|
step,
|
|
2915
3107
|
toolNames: resolveDagPiToolNames(input.task),
|
|
@@ -2922,7 +3114,6 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
2922
3114
|
...(input.task.contextBudget
|
|
2923
3115
|
? { contextBudget: input.task.contextBudget }
|
|
2924
3116
|
: {}),
|
|
2925
|
-
...(writerToolPolicy ? { writerToolPolicy } : {}),
|
|
2926
3117
|
...(piExtensionPaths && piExtensionPaths.length > 0
|
|
2927
3118
|
? { piExtensionPaths }
|
|
2928
3119
|
: {}),
|
|
@@ -2935,7 +3126,55 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
2935
3126
|
bridgeActivity(activity.kind, activity.at);
|
|
2936
3127
|
}
|
|
2937
3128
|
: undefined,
|
|
2938
|
-
}
|
|
3129
|
+
};
|
|
3130
|
+
if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
|
|
3131
|
+
// Frontend-only split: three sequential sessions with independent
|
|
3132
|
+
// output budgets (coverage -> UX decisions -> finalize), mirroring the
|
|
3133
|
+
// backend-test template's module sharding. Every other template keeps
|
|
3134
|
+
// the single-session path below.
|
|
3135
|
+
let planRequirementIds = [];
|
|
3136
|
+
try {
|
|
3137
|
+
const { readCommittedOriginFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
|
|
3138
|
+
const contractFacts = await readCommittedOriginFacts(meta.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl");
|
|
3139
|
+
// Only requirement facts: the contract ledger also carries
|
|
3140
|
+
// constraints (CON-*), evidence expectations (EV-*), handoff
|
|
3141
|
+
// intents (HND-*), open questions (OQ-*) and split proposals
|
|
3142
|
+
// (SPLIT-*) that all have ids — feeding those into the coverage
|
|
3143
|
+
// batches made the model record non-frozen plan-requirement ids
|
|
3144
|
+
// that finalize's canonical-coverage gate then rejected (r-ext2).
|
|
3145
|
+
planRequirementIds = contractFacts
|
|
3146
|
+
.filter((record) => record.fact
|
|
3147
|
+
?.kind === "requirement")
|
|
3148
|
+
.map((record) => record.fact?.id)
|
|
3149
|
+
.filter((id) => typeof id === "string")
|
|
3150
|
+
.sort();
|
|
3151
|
+
}
|
|
3152
|
+
catch {
|
|
3153
|
+
// Unreadable ledger falls back to a single coverage session.
|
|
3154
|
+
}
|
|
3155
|
+
result = await runFrontendPlanSegmentedSessions({
|
|
3156
|
+
piStepFn,
|
|
3157
|
+
sessionOptions: piSessionOptions,
|
|
3158
|
+
basePrompt: input.prompt,
|
|
3159
|
+
attempt: input.attempt ?? 1,
|
|
3160
|
+
committedFactCount: () => planLedgerTools.committedFactCount(),
|
|
3161
|
+
requirementIds: planRequirementIds,
|
|
3162
|
+
committedRequirementIds: () => planLedgerTools.committedRequirementIds(),
|
|
3163
|
+
segmentCustomTools: (toolNames) => toolNames === null
|
|
3164
|
+
? planLedgerTools.customTools
|
|
3165
|
+
: planLedgerTools.customTools.filter((tool) => typeof tool === "object" &&
|
|
3166
|
+
tool !== null &&
|
|
3167
|
+
toolNames.has(tool.name)),
|
|
3168
|
+
flushLedger: () => planLedgerTools.flush(),
|
|
3169
|
+
});
|
|
3170
|
+
}
|
|
3171
|
+
else {
|
|
3172
|
+
result = await piStepFn({
|
|
3173
|
+
...piSessionOptions,
|
|
3174
|
+
prompt: input.prompt,
|
|
3175
|
+
...(writerToolPolicy ? { writerToolPolicy } : {}),
|
|
3176
|
+
});
|
|
3177
|
+
}
|
|
2939
3178
|
}
|
|
2940
3179
|
catch (error) {
|
|
2941
3180
|
if (playwrightToolContext) {
|
|
@@ -3072,25 +3311,10 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
3072
3311
|
catch {
|
|
3073
3312
|
// best-effort flush; missing ledger still fails at the node validator
|
|
3074
3313
|
}
|
|
3075
|
-
//
|
|
3076
|
-
//
|
|
3077
|
-
//
|
|
3078
|
-
//
|
|
3079
|
-
// session log and retry with a reduced-reading instruction.
|
|
3080
|
-
const readBurstIssues = input.task.readBudget?.onExhaustion === "return-guidance"
|
|
3081
|
-
? []
|
|
3082
|
-
: await detectPlanReadBurst({
|
|
3083
|
-
runDir: meta.runDir,
|
|
3084
|
-
nodeId: input.task.id,
|
|
3085
|
-
});
|
|
3086
|
-
if (readBurstIssues.length > 0) {
|
|
3087
|
-
return {
|
|
3088
|
-
...mapped,
|
|
3089
|
-
ok: false,
|
|
3090
|
-
failureCategory: "read-burst",
|
|
3091
|
-
stderr: [mapped.stderr, ...readBurstIssues].filter(Boolean).join("\n\n"),
|
|
3092
|
-
};
|
|
3093
|
-
}
|
|
3314
|
+
// NOTE: the legacy plan read-burst guard lived here. It is dead code
|
|
3315
|
+
// since the plan node went tool-only (resolveDagPiToolNames grants no
|
|
3316
|
+
// read/grep/ls/find), so a read burst is structurally impossible; the
|
|
3317
|
+
// read-budget path (scout etc.) keeps its own guard.
|
|
3094
3318
|
}
|
|
3095
3319
|
if (!isWriteTask) {
|
|
3096
3320
|
if (!mapped.ok &&
|
|
@@ -539,6 +539,18 @@ export const frontendImplementationContractSchema = z
|
|
|
539
539
|
message: `unknown verification target ${targetId}`,
|
|
540
540
|
path: ["requirements"],
|
|
541
541
|
});
|
|
542
|
+
// Interactions bind to VTs too: an interaction whose behavior
|
|
543
|
+
// verification points at an unmaterialized VT is untraceable —
|
|
544
|
+
// r20 review finding (dangling *-BEHAVIOR references sailed
|
|
545
|
+
// through plan/design/implement to the final review).
|
|
546
|
+
for (const interaction of value.interactions)
|
|
547
|
+
for (const targetId of interaction.verificationTargetIds)
|
|
548
|
+
if (!verificationIds.includes(targetId))
|
|
549
|
+
ctx.addIssue({
|
|
550
|
+
code: "custom",
|
|
551
|
+
message: `interaction "${interaction.name}" references unknown verification target ${targetId}`,
|
|
552
|
+
path: ["interactions"],
|
|
553
|
+
});
|
|
542
554
|
// Source fidelity ledger binding (AC-005/AC-006): 绑定携带
|
|
543
555
|
// requirementToFragments(ledger v2)时,每个 requirement 必须携带
|
|
544
556
|
// 非空 sourceFragmentIds,否则 fail closed(防引用伪造/缺失)。
|
|
@@ -570,12 +582,23 @@ export const frontendImplementationContractSchema = z
|
|
|
570
582
|
if (state.applicable &&
|
|
571
583
|
(!state.expectedBehavior ||
|
|
572
584
|
!state.implementationTargets?.length ||
|
|
573
|
-
!state.verificationTargetIds?.length))
|
|
585
|
+
!state.verificationTargetIds?.length)) {
|
|
586
|
+
// Name the state and the exact missing fields: the fixer is a
|
|
587
|
+
// model iterating on finalize receipts — it cannot fix a
|
|
588
|
+
// defect it cannot locate (r18: 39 blind finalize retries).
|
|
589
|
+
const missing = [];
|
|
590
|
+
if (!state.expectedBehavior)
|
|
591
|
+
missing.push("expectedBehavior");
|
|
592
|
+
if (!state.implementationTargets?.length)
|
|
593
|
+
missing.push("implementationTargets");
|
|
594
|
+
if (!state.verificationTargetIds?.length)
|
|
595
|
+
missing.push("verificationTargetIds");
|
|
574
596
|
ctx.addIssue({
|
|
575
597
|
code: "custom",
|
|
576
|
-
message:
|
|
598
|
+
message: `applicable UI state "${state.name}" is missing: ${missing.join(", ")} — record_state_flow it again with those fields filled`,
|
|
577
599
|
path: ["uiStates"],
|
|
578
600
|
});
|
|
601
|
+
}
|
|
579
602
|
if (!state.applicable && !state.notApplicableReason)
|
|
580
603
|
ctx.addIssue({
|
|
581
604
|
code: "custom",
|
|
@@ -1826,12 +1849,25 @@ export async function analyzeFrontendImplementationContract(input) {
|
|
|
1826
1849
|
if (parsedTargetFiles.some((file) => file.startsWith("/") || file.includes("\\")))
|
|
1827
1850
|
fail("blocked", "invalid-output: frontend contract target paths must be relative POSIX paths", candidateJsonSha256);
|
|
1828
1851
|
const parsedStates = asRecord(parsed)?.uiStates;
|
|
1829
|
-
if (Array.isArray(parsedStates)
|
|
1830
|
-
const
|
|
1831
|
-
|
|
1832
|
-
|
|
1833
|
-
|
|
1834
|
-
|
|
1852
|
+
if (Array.isArray(parsedStates)) {
|
|
1853
|
+
const incompleteStates = [];
|
|
1854
|
+
for (const item of parsedStates) {
|
|
1855
|
+
const state = asRecord(item);
|
|
1856
|
+
if (!state || state.applicable !== true)
|
|
1857
|
+
continue;
|
|
1858
|
+
const missing = [];
|
|
1859
|
+
if (!asString(state.expectedBehavior))
|
|
1860
|
+
missing.push("expectedBehavior");
|
|
1861
|
+
if (asStringArray(state.implementationTargets).length === 0)
|
|
1862
|
+
missing.push("implementationTargets");
|
|
1863
|
+
if (asStringArray(state.verificationTargetIds).length === 0)
|
|
1864
|
+
missing.push("verificationTargetIds");
|
|
1865
|
+
if (missing.length > 0)
|
|
1866
|
+
incompleteStates.push(`"${asString(state.name)}": missing ${missing.join(", ")}`);
|
|
1867
|
+
}
|
|
1868
|
+
if (incompleteStates.length > 0)
|
|
1869
|
+
fail("retryable-invalid", `invalid-output: applicable UI states incomplete — re-record each with record_state_flow filling the named fields: ${incompleteStates.join("; ")}`, candidateJsonSha256);
|
|
1870
|
+
}
|
|
1835
1871
|
const parsedMockApi = asRecord(parsed)?.mockApi;
|
|
1836
1872
|
if (asRecord(parsedMockApi) &&
|
|
1837
1873
|
typeof asRecord(parsedMockApi)?.strategy === "string" &&
|
|
@@ -2079,7 +2115,7 @@ const VERIFICATION_SYMBOL_MAX_CHARS = 60;
|
|
|
2079
2115
|
* loading") and `describe(...)` / `it(...)` forms stay valid — a real symbol
|
|
2080
2116
|
* may be a function name, a dotted path, or a describe/it title.
|
|
2081
2117
|
*/
|
|
2082
|
-
function isSuspiciousVerificationSymbol(symbol) {
|
|
2118
|
+
export function isSuspiciousVerificationSymbol(symbol) {
|
|
2083
2119
|
const trimmed = symbol.trim();
|
|
2084
2120
|
if (!trimmed)
|
|
2085
2121
|
return false;
|
|
@@ -2126,6 +2162,17 @@ export class PlanPolicyPrecheckFailure extends Error {
|
|
|
2126
2162
|
*/
|
|
2127
2163
|
export async function analyzeFrontendPlanPatchCandidate(input) {
|
|
2128
2164
|
const analysis = await analyzeFrontendImplementationContract(input);
|
|
2165
|
+
// A committed VT with a fabricated symbol is an immutable ledger fact —
|
|
2166
|
+
// the record boundary rejects duplicate ids, so the model cannot overwrite
|
|
2167
|
+
// it and throwing here deadlocks the receipt loop (r19: VT-AC006-BEHAVIOR).
|
|
2168
|
+
// Drop suspicious symbols deterministically instead: the VT stays valid and
|
|
2169
|
+
// the deterministic trace gate verifies file+command (and resolvability)
|
|
2170
|
+
// after verification. Mirrors the shell materialization's drop semantics.
|
|
2171
|
+
for (const target of analysis.canonical.verificationTargets) {
|
|
2172
|
+
if (target.symbol && isSuspiciousVerificationSymbol(target.symbol)) {
|
|
2173
|
+
target.symbol = undefined;
|
|
2174
|
+
}
|
|
2175
|
+
}
|
|
2129
2176
|
// Front-load the verification-symbol shape check so fabricated symbols
|
|
2130
2177
|
// are fixed by the plan retry ladder in-node instead of failing the
|
|
2131
2178
|
// verify trace gate at the end of the run.
|
|
@@ -287,9 +287,20 @@ export async function runFrontendReviewContextGate(input) {
|
|
|
287
287
|
if (!parsedContract.success) {
|
|
288
288
|
throw new FrontendReviewContextFailure("review-context-invalid-contract", "frontend review context invalid implementation contract");
|
|
289
289
|
}
|
|
290
|
+
// Authorized diff surface = deliverable files + every verification-target
|
|
291
|
+
// file the contract references. The writer legitimately writes VT test
|
|
292
|
+
// artifacts (behavior verification), and the reviewer must audit their
|
|
293
|
+
// diffs; scoping this to targets.files alone failed the baseline-overlap
|
|
294
|
+
// check on app.test.js (r19 extreme smoke).
|
|
295
|
+
const authorizedChangedPaths = [
|
|
296
|
+
...new Set([
|
|
297
|
+
...parsedContract.data.targets.files,
|
|
298
|
+
...parsedContract.data.verificationTargets.map((target) => target.file),
|
|
299
|
+
]),
|
|
300
|
+
];
|
|
290
301
|
const diff = await runFrontendWorktreeDiffGate({
|
|
291
302
|
...input,
|
|
292
|
-
authorizedChangedPaths
|
|
303
|
+
authorizedChangedPaths,
|
|
293
304
|
});
|
|
294
305
|
const verificationTrace = await readRequiredJson(input.runDir, "contracts/frontend-verification-trace.json");
|
|
295
306
|
const repairAssessment = await readOptionalRepairAssessment(input.runDir);
|
|
@@ -260,7 +260,12 @@ export function restorePlanPatchFromCommittedFacts(records) {
|
|
|
260
260
|
if (isRecord(fact.patch))
|
|
261
261
|
return fact.patch;
|
|
262
262
|
}
|
|
263
|
-
|
|
263
|
+
// No finalize-published snapshot: the receipt fix loop can end an attempt
|
|
264
|
+
// before any finalize_plan succeeds while dozens of record_* facts are
|
|
265
|
+
// already committed. Assemble the patch from those record facts instead of
|
|
266
|
+
// declaring the ledger missing (r19: "ledger missing" discarded a ledger
|
|
267
|
+
// with 92 committed facts).
|
|
268
|
+
return assemblePlanPatchFromCommittedFacts(records);
|
|
264
269
|
}
|
|
265
270
|
/**
|
|
266
271
|
* A+B (AC-004): reverse of `planLedgerFactsFromPatch` for the incremental
|
|
@@ -280,12 +285,33 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
|
|
|
280
285
|
PLAN_LEDGER_FACT_KINDS.includes(fact.kind));
|
|
281
286
|
if (facts.length === 0)
|
|
282
287
|
return undefined;
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
288
|
+
// Singleton record facts: the LAST committed fact wins. The model corrects
|
|
289
|
+
// a rejected plan by re-recording the fact (r18: 39 finalize retries never
|
|
290
|
+
// converged because the first state-flow fact kept shadowing the
|
|
291
|
+
// corrections). Requirements / verification targets / evidence gaps stay
|
|
292
|
+
// additive (aggregated below); component choices accumulate; singletons
|
|
293
|
+
// supersede.
|
|
294
|
+
const lastByKind = (kind) => {
|
|
295
|
+
let found;
|
|
296
|
+
for (const fact of facts) {
|
|
297
|
+
if (fact.kind === kind)
|
|
298
|
+
found = fact;
|
|
299
|
+
}
|
|
300
|
+
return found;
|
|
301
|
+
};
|
|
302
|
+
// Patch snapshots (published by earlier finalize calls) carry no routes;
|
|
303
|
+
// exclude them so a snapshot cannot shadow the model's route selection.
|
|
304
|
+
const routeSurface = (() => {
|
|
305
|
+
let found;
|
|
306
|
+
for (const fact of facts) {
|
|
307
|
+
if (fact.kind !== "target-surface" || isRecord(fact.patch))
|
|
308
|
+
continue;
|
|
309
|
+
found = fact;
|
|
310
|
+
}
|
|
311
|
+
return found;
|
|
312
|
+
})();
|
|
287
313
|
const patch = {};
|
|
288
|
-
const targetSurface =
|
|
314
|
+
const targetSurface = routeSurface;
|
|
289
315
|
if (targetSurface) {
|
|
290
316
|
const routes = canonicalStringSet(asStringArray(targetSurface.routes));
|
|
291
317
|
if (routes.length > 0)
|
|
@@ -295,14 +321,34 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
|
|
|
295
321
|
const collectedChoices = componentChoices.flatMap((fact) => Array.isArray(fact.uiComponentChoices)
|
|
296
322
|
? fact.uiComponentChoices.filter(isRecord)
|
|
297
323
|
: []);
|
|
324
|
+
// Later corrections supersede earlier submissions per purpose: without
|
|
325
|
+
// this, a re-recorded choice (fixing a wrong specReference) would appear
|
|
326
|
+
// twice and the reviewer would still audit the stale entry (r19).
|
|
327
|
+
const choicesByPurpose = new Map();
|
|
328
|
+
const dedupedChoices = [];
|
|
329
|
+
for (const choice of collectedChoices) {
|
|
330
|
+
const purpose = asString(choice.purpose);
|
|
331
|
+
if (!purpose) {
|
|
332
|
+
dedupedChoices.push(choice);
|
|
333
|
+
continue;
|
|
334
|
+
}
|
|
335
|
+
if (choicesByPurpose.has(purpose)) {
|
|
336
|
+
const at = dedupedChoices.findIndex((existing) => asString(existing.purpose) === purpose);
|
|
337
|
+
dedupedChoices[at] = choice;
|
|
338
|
+
}
|
|
339
|
+
else {
|
|
340
|
+
choicesByPurpose.set(purpose, choice);
|
|
341
|
+
dedupedChoices.push(choice);
|
|
342
|
+
}
|
|
343
|
+
}
|
|
298
344
|
const stylingStrategy = componentChoices
|
|
299
345
|
.map((fact) => asString(fact.stylingStrategy))
|
|
300
346
|
.find((value) => value.length > 0);
|
|
301
|
-
if (
|
|
302
|
-
patch.uiComponentChoices =
|
|
347
|
+
if (dedupedChoices.length > 0)
|
|
348
|
+
patch.uiComponentChoices = dedupedChoices;
|
|
303
349
|
if (stylingStrategy)
|
|
304
350
|
patch.stylingStrategy = stylingStrategy;
|
|
305
|
-
const stateFlow =
|
|
351
|
+
const stateFlow = lastByKind("state-flow");
|
|
306
352
|
if (stateFlow) {
|
|
307
353
|
if (Array.isArray(stateFlow.uiStates)) {
|
|
308
354
|
patch.uiStates = stateFlow.uiStates.filter(isRecord);
|
|
@@ -311,17 +357,17 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
|
|
|
311
357
|
patch.interactions = stateFlow.interactions.filter(isRecord);
|
|
312
358
|
}
|
|
313
359
|
}
|
|
314
|
-
const mockApi =
|
|
360
|
+
const mockApi = lastByKind("mock-api");
|
|
315
361
|
if (mockApi && isRecord(mockApi.mockApi)) {
|
|
316
362
|
patch.mockApi = mockApi.mockApi;
|
|
317
363
|
}
|
|
318
|
-
const deviation =
|
|
364
|
+
const deviation = lastByKind("design-deviation");
|
|
319
365
|
if (deviation) {
|
|
320
366
|
const conflicts = canonicalStringSet(asStringArray(deviation.conflicts));
|
|
321
367
|
if (conflicts.length > 0)
|
|
322
368
|
patch.designEvidence = { conflicts };
|
|
323
369
|
}
|
|
324
|
-
const dependency =
|
|
370
|
+
const dependency = lastByKind("dependency");
|
|
325
371
|
const dependencyPolicy = asString(dependency?.policy);
|
|
326
372
|
if (dependencyPolicy)
|
|
327
373
|
patch.dependencyPolicy = dependencyPolicy;
|
|
@@ -3210,6 +3210,7 @@ async function buildFrontendHybridDagFromTask(sources) {
|
|
|
3210
3210
|
...(requiresOpenspecClassification ? ["When a component choice uses an OpenSpec selection, cite that selection; otherwise do not classify unrelated candidates."] : []),
|
|
3211
3211
|
"Call finalize_plan exactly once after the necessary typed facts. Return no Markdown narrative.",
|
|
3212
3212
|
"TOOL-ONLY PLAN: Do not read Contract/Scout stdout, task sources, or Scout-confirmed target files. Contract and Scout already own evidence discovery; use the injected upstream facts, record a genuine evidence gap when those facts are insufficient, and start committing record_* facts immediately. For decision=new, pass sourceRequirementIds to record_component_choice; runtime derives the exact PRD citation from the frozen ledger.",
|
|
3213
|
+
"Output budget protocol (hard, max output <=16K per turn): never enumerate-reason the whole requirement list before your first record_* call — that reasoning burns the entire output budget and the attempt dies with zero committed facts. Process requirements in order: think about ONE requirement briefly, immediately emit its record calls (up to 5 per message), then move to the next. If your budget runs low, stop recording and call finalize_plan with what is committed — the retry ladder continues the remainder in a fresh session.",
|
|
3213
3214
|
fixedVerificationContext,
|
|
3214
3215
|
scopedOpenspecContext,
|
|
3215
3216
|
mockContextBlock,
|
|
@@ -315,9 +315,11 @@ export const dagFrontendVerificationBundleSchema = z
|
|
|
315
315
|
behaviorEvidence: dagShellVerifyEvidenceSchema,
|
|
316
316
|
lintBaselineNodeId: dagFrontendNodeIdSchema.optional(),
|
|
317
317
|
writerNodeIds: z.array(dagFrontendNodeIdSchema).optional(),
|
|
318
|
-
/**
|
|
319
|
-
*
|
|
320
|
-
|
|
318
|
+
/** "initial" = first-pass assess bundle; "repair" = the convergence
|
|
319
|
+
* loop's post-repair reverify, which re-runs the same frozen commands
|
|
320
|
+
* and fails on any failure (still no same-run repair branch inside the
|
|
321
|
+
* node itself). */
|
|
322
|
+
mode: z.enum(["initial", "repair"]),
|
|
321
323
|
})
|
|
322
324
|
.superRefine((bundle, context) => {
|
|
323
325
|
const groups = [
|
package/harness.json
CHANGED
|
@@ -83,8 +83,8 @@
|
|
|
83
83
|
"pi": {
|
|
84
84
|
"description": "Pi 负责规划、评审、诊断;当 DAG toolProfile=write 时也可做有界写入。模型按复杂度三档配置,支持 provider/model 字符串或带 thinking 的对象;斜杠前为 Pi provider,后为 modelId,勿只写裸 modelId。",
|
|
85
85
|
"LOW": "wizard-local/minimax-m3",
|
|
86
|
-
"MED": "wizard-local/gpt-5.
|
|
87
|
-
"HIGH": "wizard-local/gpt-5.
|
|
86
|
+
"MED": "wizard-local/gpt-5.5",
|
|
87
|
+
"HIGH": "wizard-local/gpt-5.5"
|
|
88
88
|
}
|
|
89
89
|
}
|
|
90
|
-
}
|
|
90
|
+
}
|