@tea-agent/loop-agent 0.39.0-beta.13 → 0.39.0-beta.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +2 -0
- package/dist/application/dag/generate-task-dag.js +26 -2
- package/dist/application/task-lifecycle/advance.js +20 -4
- package/dist/application/task-lifecycle/observe.js +171 -17
- package/dist/application/task-lifecycle/plan-transitions.js +42 -7
- package/dist/build-stamp.json +3 -3
- package/dist/executors/dag-pi-executor.js +308 -84
- package/dist/task/contract/adopt.js +4 -0
- package/dist/task/contract/import-revision.js +4 -0
- package/dist/workflows/dag/frontend-implementation-contract.js +56 -9
- package/dist/workflows/dag/frontend-review-context.js +12 -1
- package/dist/workflows/dag/frontend-shadow-dual-write.js +58 -12
- package/dist/workflows/dag/init-hybrid.js +1 -0
- package/dist/workflows/dag/types.js +5 -3
- package/harness.json +3 -3
- package/package.json +2 -2
|
@@ -16,6 +16,7 @@ import { redactPromptForLog, truncateOutput, } from "../shared/output-truncation
|
|
|
16
16
|
import { GitStatusUnavailableError, pathsChangedDuringRun, readGitStatusPorcelain, recoverRootNulArtifact, snapshotGitStatusPathFingerprints, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
|
|
17
17
|
import { captureWorkspaceWriteSnapshot, diffWorkspaceWriteSnapshots, } from "./workspace-write-snapshot.js";
|
|
18
18
|
import { pathMatchesPattern } from "../shared/git-progress.js";
|
|
19
|
+
import { isSuspiciousVerificationSymbol } from "../workflows/dag/frontend-implementation-contract.js";
|
|
19
20
|
import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
|
|
20
21
|
import { assessBackendTestMdPlanCompleteness, assessBackendTestMdWriterCompleteness, assessBackendTestPytestPlanCompleteness, assessBackendTestPytestWriterCompleteness, assessBackendTestShardChildCompleteness, backendTestWriterProgressRoleForTask, classifyBackendTestWriterCompletenessFailure, isBackendTestCompletenessRetryCandidate, isBackendTestMdPlanTask, isBackendTestPytestPlanTask, isBackendTestShardChildTask, writeBackendTestWriterProgressArtifacts, } from "../workflows/dag/backend-test-writer-completeness.js";
|
|
21
22
|
import { resolveBackendTestLayout } from "../workflows/dag/backend-test-layout.js";
|
|
@@ -236,16 +237,6 @@ export const FRONTEND_PLAN_RECORD_TOOL_NAMES = [
|
|
|
236
237
|
];
|
|
237
238
|
export const FRONTEND_PLAN_TERMINAL_TOOL_NAMES = ["finalize_plan"];
|
|
238
239
|
export const FRONTEND_PLAN_ADOPT_TOOL_NAMES = ["adopt_staged_fact"];
|
|
239
|
-
/**
|
|
240
|
-
* Inter-call delay for high-frequency frontend record_* tools. Incremental
|
|
241
|
-
* submission (one model response per 1-5 entries) means dozens of sequential
|
|
242
|
-
* API round trips in a few minutes; third-party gateways (huu.dqy.ink) trip
|
|
243
|
-
* rate limits whose windows exceed normal API RPM even at ~13 calls/min. A
|
|
244
|
-
* short fixed delay before each tool body keeps the sustained rate under
|
|
245
|
-
* typical thresholds while batching cuts the total call count.
|
|
246
|
-
*/
|
|
247
|
-
const FRONTEND_RECORD_TOOL_THROTTLE_MS = 1000;
|
|
248
|
-
const sleepRecordThrottle = () => new Promise((resolve) => setTimeout(resolve, FRONTEND_RECORD_TOOL_THROTTLE_MS));
|
|
249
240
|
/** M5: `frontend-review-pi` emits its authoritative terminal verdict through
|
|
250
241
|
* committed typed tools instead of the legacy JSON verdict parse. */
|
|
251
242
|
export function isFrontendReviewTypedTerminalNode(task) {
|
|
@@ -395,10 +386,6 @@ export function scanReviewTerminalKindsFromSessionEvents(content) {
|
|
|
395
386
|
}
|
|
396
387
|
return kinds;
|
|
397
388
|
}
|
|
398
|
-
/** Read-only discovery tool budget for frontend-plan-pi. Exceeding it means
|
|
399
|
-
* the plan re-read upstream outputs/sources instead of trusting typed facts,
|
|
400
|
-
* which blows up the context window (400 request-too-large). */
|
|
401
|
-
const PLAN_READ_TOOL_BUDGET = 40;
|
|
402
389
|
const READ_ONLY_TOOL_NAMES = new Set(["read", "grep", "ls", "find"]);
|
|
403
390
|
function readEventPath(event) {
|
|
404
391
|
const candidates = [event.path, event.readPath, event.input];
|
|
@@ -483,45 +470,6 @@ export async function detectNodeReadBudget(input) {
|
|
|
483
470
|
issues.push(`frontend ${input.nodeId} read budget exceeded: ${stats.elapsedMs}ms (budget ${input.budget.maxMs}ms)`);
|
|
484
471
|
return issues;
|
|
485
472
|
}
|
|
486
|
-
/** Deterministic read-burst guard for frontend-plan-pi: count read-only
|
|
487
|
-
* discovery tool calls (read/grep/ls/find) from the session log. Over budget →
|
|
488
|
-
* read-burst, retried with a reduced-reading instruction. Pure scan; a
|
|
489
|
-
* successful plan under budget is never blocked. */
|
|
490
|
-
export async function detectPlanReadBurst(input) {
|
|
491
|
-
const sessionEventsPath = path.join(input.runDir, input.nodeId, "session-events.jsonl");
|
|
492
|
-
let count = 0;
|
|
493
|
-
try {
|
|
494
|
-
const content = await readFile(sessionEventsPath, "utf8");
|
|
495
|
-
for (const line of content.split("\n")) {
|
|
496
|
-
if (!line.trim())
|
|
497
|
-
continue;
|
|
498
|
-
try {
|
|
499
|
-
const event = JSON.parse(line);
|
|
500
|
-
if (event.type === "tool_execution_start" &&
|
|
501
|
-
typeof event.toolName === "string" &&
|
|
502
|
-
(event.toolName === "read" ||
|
|
503
|
-
event.toolName === "grep" ||
|
|
504
|
-
event.toolName === "ls" ||
|
|
505
|
-
event.toolName === "find")) {
|
|
506
|
-
count += 1;
|
|
507
|
-
}
|
|
508
|
-
}
|
|
509
|
-
catch {
|
|
510
|
-
// skip unparseable line
|
|
511
|
-
}
|
|
512
|
-
}
|
|
513
|
-
}
|
|
514
|
-
catch {
|
|
515
|
-
// Missing/unreadable session log → no burst detection
|
|
516
|
-
return [];
|
|
517
|
-
}
|
|
518
|
-
if (count > PLAN_READ_TOOL_BUDGET) {
|
|
519
|
-
return [
|
|
520
|
-
`frontend plan read-burst: ${count} read-only tool calls (budget ${PLAN_READ_TOOL_BUDGET}). Trust the upstream contract/scout typed facts; do not re-read contract/scout outputs or source files already captured. Minimize discovery reads, commit record_* facts directly, then finalize_plan.`,
|
|
521
|
-
];
|
|
522
|
-
}
|
|
523
|
-
return [];
|
|
524
|
-
}
|
|
525
473
|
export const FRONTEND_DESIGN_TERMINAL_TOOL_NAMES = new Set([
|
|
526
474
|
"approve_design",
|
|
527
475
|
"request_design_changes",
|
|
@@ -932,11 +880,27 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
932
880
|
import("typebox"),
|
|
933
881
|
import("@earendil-works/pi-coding-agent"),
|
|
934
882
|
]);
|
|
935
|
-
const { readCommittedEvents, writeTypedEventStoreJsonl } = await import("../workflows/dag/frontend-typed-event-store.js");
|
|
883
|
+
const { loadTypedEventStore, readCommittedEvents, writeTypedEventStoreJsonl, } = await import("../workflows/dag/frontend-typed-event-store.js");
|
|
936
884
|
const { adoptStagedFact, adoptTypedEventFact, stageTypedEventFact } = await import("../workflows/dag/frontend-typed-event-transaction.js");
|
|
937
885
|
const { assemblePlanPatchFromCommittedFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
|
|
938
886
|
const store = input.store;
|
|
939
887
|
const attemptId = input.attemptId;
|
|
888
|
+
// A retry creates a fresh executor-local store, but the plan ledger is the
|
|
889
|
+
// cross-attempt authority. Restore the committed prefix before registering
|
|
890
|
+
// tools; otherwise the first flush of a retry can overwrite facts that the
|
|
891
|
+
// previous attempt had already committed. The on-disk file contains only
|
|
892
|
+
// committed records, so loading it is also fail-closed with respect to
|
|
893
|
+
// staged/quarantined facts.
|
|
894
|
+
const persisted = await loadTypedEventStore(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"));
|
|
895
|
+
if (persisted.records.length > 0) {
|
|
896
|
+
const existingEventIds = new Set(store.records.map((record) => record.eventId));
|
|
897
|
+
for (const record of persisted.records) {
|
|
898
|
+
if (!existingEventIds.has(record.eventId)) {
|
|
899
|
+
store.records.push(record);
|
|
900
|
+
}
|
|
901
|
+
}
|
|
902
|
+
store.revision = Math.max(store.revision, persisted.revision);
|
|
903
|
+
}
|
|
940
904
|
const stringArray = Type.Array(Type.String({}));
|
|
941
905
|
const optionalString = Type.Optional(Type.String({}));
|
|
942
906
|
const optionalStringArray = Type.Optional(stringArray);
|
|
@@ -1127,7 +1091,6 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1127
1091
|
stylingStrategy: optionalString,
|
|
1128
1092
|
}, { additionalProperties: false }),
|
|
1129
1093
|
async execute(_toolCallId, params) {
|
|
1130
|
-
await sleepRecordThrottle();
|
|
1131
1094
|
const rawChoice = params?.choice;
|
|
1132
1095
|
if (!isRecordObject(rawChoice)) {
|
|
1133
1096
|
return planToolReceipt({
|
|
@@ -1180,7 +1143,16 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1180
1143
|
? { stylingStrategy: params.stylingStrategy }
|
|
1181
1144
|
: {}),
|
|
1182
1145
|
});
|
|
1183
|
-
|
|
1146
|
+
// Echo the frozen citation the runtime derived: the model sees the
|
|
1147
|
+
// purpose↔citation mapping it just committed and can re-record the
|
|
1148
|
+
// choice (last-wins per purpose at compile) when it mismatches.
|
|
1149
|
+
const echo = {
|
|
1150
|
+
...result,
|
|
1151
|
+
...(choice.specReference
|
|
1152
|
+
? { derivedSpecReference: choice.specReference }
|
|
1153
|
+
: {}),
|
|
1154
|
+
};
|
|
1155
|
+
return planToolReceipt(echo);
|
|
1184
1156
|
},
|
|
1185
1157
|
});
|
|
1186
1158
|
const recordStateFlowTool = defineTool({
|
|
@@ -1264,6 +1236,26 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1264
1236
|
error: "record_state_flow requires non-empty interaction.name (id is accepted only as a legacy alias)",
|
|
1265
1237
|
});
|
|
1266
1238
|
}
|
|
1239
|
+
// Interaction -> VT forward references are legal only against
|
|
1240
|
+
// already-committed VT facts. In the segmented flow every VT
|
|
1241
|
+
// commits in the coverage segment before state flows run, so a
|
|
1242
|
+
// dangling reference here is a real defect (r20: *-BEHAVIOR
|
|
1243
|
+
// refs reached the final review untraceable).
|
|
1244
|
+
const interactionVtIds = stringList(interaction.verificationTargetIds);
|
|
1245
|
+
const committedVtIds = new Set(readCommittedEvents(store, attemptId)
|
|
1246
|
+
.map((event) => event.fact)
|
|
1247
|
+
.filter((fact) => fact.kind === "plan-verification-target")
|
|
1248
|
+
.map((fact) => fact.entry
|
|
1249
|
+
?.id)
|
|
1250
|
+
.filter((id) => typeof id === "string"));
|
|
1251
|
+
const unknownVtIds = interactionVtIds.filter((id) => !committedVtIds.has(id));
|
|
1252
|
+
if (unknownVtIds.length > 0) {
|
|
1253
|
+
return planToolReceipt({
|
|
1254
|
+
ok: false,
|
|
1255
|
+
kind: "state-flow",
|
|
1256
|
+
error: `record_state_flow interaction "${resolvedName}" references verification targets that are not recorded yet: ${unknownVtIds.join(", ")}; record them with record_plan_verification_target first, then re-record this state flow`,
|
|
1257
|
+
});
|
|
1258
|
+
}
|
|
1267
1259
|
interactions.push({ ...interaction, name: resolvedName });
|
|
1268
1260
|
}
|
|
1269
1261
|
const states = uiStates
|
|
@@ -1359,7 +1351,6 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1359
1351
|
entry: requirementSchema,
|
|
1360
1352
|
}, { additionalProperties: false }),
|
|
1361
1353
|
async execute(_toolCallId, params) {
|
|
1362
|
-
await sleepRecordThrottle();
|
|
1363
1354
|
const rawEntry = params?.entry;
|
|
1364
1355
|
if (!isRecordObject(rawEntry)) {
|
|
1365
1356
|
return planToolReceipt({
|
|
@@ -1387,11 +1378,26 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1387
1378
|
void _omittedGap;
|
|
1388
1379
|
entry = rest;
|
|
1389
1380
|
}
|
|
1381
|
+
// Canonical-identity check: a requirement id outside the frozen
|
|
1382
|
+
// canonical list (e.g. a BR-* business rule picked up from the PRD
|
|
1383
|
+
// prose) would commit an immutable fact that finalize's
|
|
1384
|
+
// canonical-coverage gate rejects with no in-node cure. Reject here
|
|
1385
|
+
// and name the allowed ids.
|
|
1386
|
+
const id = typeof entry.id === "string" ? entry.id : "";
|
|
1387
|
+
if (id &&
|
|
1388
|
+
input.requirementIds &&
|
|
1389
|
+
input.requirementIds.length > 0 &&
|
|
1390
|
+
!input.requirementIds.includes(id)) {
|
|
1391
|
+
return planToolReceipt({
|
|
1392
|
+
ok: false,
|
|
1393
|
+
kind: "plan-requirement",
|
|
1394
|
+
error: `record_plan_requirement id "${id}" is not a frozen canonical requirement; canonical ids are: ${input.requirementIds.join(", ")}`,
|
|
1395
|
+
});
|
|
1396
|
+
}
|
|
1390
1397
|
// A requirement id is a canonical identity: recording it twice would
|
|
1391
1398
|
// compile a duplicate requirements[] entry and fail design review.
|
|
1392
1399
|
// Reject duplicates at the tool boundary so the model can fix them
|
|
1393
1400
|
// in-node instead of burning the attempt on a later validation error.
|
|
1394
|
-
const id = typeof entry.id === "string" ? entry.id : "";
|
|
1395
1401
|
if (id) {
|
|
1396
1402
|
const existing = readCommittedEvents(store, attemptId).find((event) => {
|
|
1397
1403
|
const fact = event.fact;
|
|
@@ -1422,7 +1428,6 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1422
1428
|
entry: verificationTargetSchema,
|
|
1423
1429
|
}, { additionalProperties: false }),
|
|
1424
1430
|
async execute(_toolCallId, params) {
|
|
1425
|
-
await sleepRecordThrottle();
|
|
1426
1431
|
const rawEntry = params?.entry;
|
|
1427
1432
|
if (!isRecordObject(rawEntry)) {
|
|
1428
1433
|
return planToolReceipt({
|
|
@@ -1477,6 +1482,18 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1477
1482
|
error: `record_plan_verification_target file is outside the writeSet patterns [${input.writeSetPatterns.join(", ")}]: ${rawEntry.file}; verification targets must point inside the task writeSet`,
|
|
1478
1483
|
});
|
|
1479
1484
|
}
|
|
1485
|
+
// Symbol shape check at the boundary: a fabricated symbol committed
|
|
1486
|
+
// here is immutable (duplicate ids are rejected), and while the
|
|
1487
|
+
// compile drops it, catching it now lets the model fix the target
|
|
1488
|
+
// in one receipt-free step.
|
|
1489
|
+
if (typeof rawEntry.symbol === "string" &&
|
|
1490
|
+
isSuspiciousVerificationSymbol(rawEntry.symbol)) {
|
|
1491
|
+
return planToolReceipt({
|
|
1492
|
+
ok: false,
|
|
1493
|
+
kind: "plan-verification-target",
|
|
1494
|
+
error: `record_plan_verification_target symbol "${rawEntry.symbol}" looks fabricated; use a real exported/describe/it symbol from ${rawEntry.file ?? "the target file"} or omit the symbol entirely (the trace gate verifies file+command)`,
|
|
1495
|
+
});
|
|
1496
|
+
}
|
|
1480
1497
|
// uiStates: [] means this verification target is intentionally not
|
|
1481
1498
|
// bound to a named UI state. Keep that canonical representation even
|
|
1482
1499
|
// when a model omits the optional tool-boundary field.
|
|
@@ -1539,7 +1556,6 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1539
1556
|
entry: evidenceGapSchema,
|
|
1540
1557
|
}, { additionalProperties: false }),
|
|
1541
1558
|
async execute(_toolCallId, params) {
|
|
1542
|
-
await sleepRecordThrottle();
|
|
1543
1559
|
const rawEntry = params?.entry;
|
|
1544
1560
|
if (!isRecordObject(rawEntry)) {
|
|
1545
1561
|
return planToolReceipt({
|
|
@@ -1732,6 +1748,19 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1732
1748
|
const committed = readCommittedEvents(store, attemptId);
|
|
1733
1749
|
await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "plan-typed-facts.jsonl"), committed);
|
|
1734
1750
|
},
|
|
1751
|
+
committedFactCount: () => readCommittedEvents(store, attemptId).length,
|
|
1752
|
+
committedRequirementIds: () => {
|
|
1753
|
+
const ids = new Set();
|
|
1754
|
+
for (const event of readCommittedEvents(store, attemptId)) {
|
|
1755
|
+
const fact = event.fact;
|
|
1756
|
+
if (!fact || fact.kind !== "plan-requirement")
|
|
1757
|
+
continue;
|
|
1758
|
+
const entryFact = fact.entry;
|
|
1759
|
+
if (typeof entryFact?.id === "string")
|
|
1760
|
+
ids.add(entryFact.id);
|
|
1761
|
+
}
|
|
1762
|
+
return ids;
|
|
1763
|
+
},
|
|
1735
1764
|
};
|
|
1736
1765
|
}
|
|
1737
1766
|
function isRecordObject(value) {
|
|
@@ -1869,7 +1898,6 @@ export async function createFrontendContractTools(input) {
|
|
|
1869
1898
|
promptSnippet: `Commit 1-5 origin=contract ${kind} facts (up to 5 per message).`,
|
|
1870
1899
|
parameters: Type.Object({}, { additionalProperties: true }),
|
|
1871
1900
|
async execute(_toolCallId, params) {
|
|
1872
|
-
await sleepRecordThrottle();
|
|
1873
1901
|
const result = await adoptContractFact(kind, {
|
|
1874
1902
|
kind,
|
|
1875
1903
|
origin: "contract",
|
|
@@ -2524,6 +2552,171 @@ async function runFrontendDesignTerminalShadow(input) {
|
|
|
2524
2552
|
}
|
|
2525
2553
|
return input.mapped;
|
|
2526
2554
|
}
|
|
2555
|
+
/** Tool subsets for the three frontend plan sessions (r17 split design). */
|
|
2556
|
+
const FRONTEND_PLAN_SEGMENTS = [
|
|
2557
|
+
{
|
|
2558
|
+
id: "coverage",
|
|
2559
|
+
toolNames: new Set([
|
|
2560
|
+
"record_plan_requirement",
|
|
2561
|
+
"record_plan_verification_target",
|
|
2562
|
+
"adopt_staged_fact",
|
|
2563
|
+
]),
|
|
2564
|
+
instruction: [
|
|
2565
|
+
"PLAN SEGMENT 1/3 — coverage mapping only.",
|
|
2566
|
+
"Your ONLY job: for every frozen requirement, emit record_plan_requirement (requirement → implementation files) and record_plan_verification_target (verification target bound to requirement ids and files). Do NOT record components, UI states, mock, dependency, or routes — a follow-up session owns those.",
|
|
2567
|
+
"Do not call finalize_plan; it is not available in this segment.",
|
|
2568
|
+
].join(" "),
|
|
2569
|
+
},
|
|
2570
|
+
{
|
|
2571
|
+
id: "ux-decisions",
|
|
2572
|
+
toolNames: new Set([
|
|
2573
|
+
"record_component_choice",
|
|
2574
|
+
"record_state_flow",
|
|
2575
|
+
"record_data_flow",
|
|
2576
|
+
"record_mock_api",
|
|
2577
|
+
"record_design_deviation",
|
|
2578
|
+
"record_dependency",
|
|
2579
|
+
"record_route_selection",
|
|
2580
|
+
"adopt_staged_fact",
|
|
2581
|
+
]),
|
|
2582
|
+
instruction: [
|
|
2583
|
+
"PLAN SEGMENT 2/3 — UX decisions.",
|
|
2584
|
+
"Requirements and verification targets are already committed in the ledger (do NOT re-record them; duplicates are rejected). Your ONLY job: record component choices (decision=new requires sourceRequirementIds per the citation table), state flows, data flow, mock strategy, design deviation, dependency policy, and route selection.",
|
|
2585
|
+
"Do not call finalize_plan; it is not available in this segment.",
|
|
2586
|
+
].join(" "),
|
|
2587
|
+
},
|
|
2588
|
+
{
|
|
2589
|
+
id: "finalize",
|
|
2590
|
+
toolNames: null,
|
|
2591
|
+
instruction: [
|
|
2592
|
+
"PLAN SEGMENT 3/3 — finalize.",
|
|
2593
|
+
"All record_* tools are available: if a finalize_plan receipt reports missing or invalid facts, fix them with the named record_* tool and call finalize_plan again. Otherwise call finalize_plan exactly once with no extra fields.",
|
|
2594
|
+
].join(" "),
|
|
2595
|
+
},
|
|
2596
|
+
];
|
|
2597
|
+
/**
|
|
2598
|
+
* Coverage-batch slicing for the frontend plan split (options 1+5): the
|
|
2599
|
+
* coverage segment becomes one session per requirement slice (default 4
|
|
2600
|
+
* requirements), so a small output budget can never be exhausted by
|
|
2601
|
+
* upfront reasoning about the whole requirement list. Zero-progress
|
|
2602
|
+
* batches are split in half and retried (option 5); single-requirement
|
|
2603
|
+
* zero-progress failures short-circuit to the retry ladder.
|
|
2604
|
+
*/
|
|
2605
|
+
const FRONTEND_PLAN_COVERAGE_BATCH_SIZE = 4;
|
|
2606
|
+
const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 32;
|
|
2607
|
+
export async function runFrontendPlanSegmentedSessions(input) {
|
|
2608
|
+
const queue = [];
|
|
2609
|
+
// An empty list means "ledger unreadable / unknown" and must fall back to
|
|
2610
|
+
// one unscoped coverage session — only a non-empty list batches.
|
|
2611
|
+
const requirementIdsProvided = input.requirementIds !== undefined && input.requirementIds.length > 0;
|
|
2612
|
+
const pending = (input.requirementIds ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
|
|
2613
|
+
for (const segment of FRONTEND_PLAN_SEGMENTS) {
|
|
2614
|
+
if (segment.id !== "coverage") {
|
|
2615
|
+
queue.push({
|
|
2616
|
+
id: segment.id,
|
|
2617
|
+
toolNames: segment.toolNames,
|
|
2618
|
+
prompt: `${input.basePrompt}\n\n${segment.instruction}`,
|
|
2619
|
+
});
|
|
2620
|
+
continue;
|
|
2621
|
+
}
|
|
2622
|
+
// No requirement list (unreadable ledger) -> one unscoped coverage
|
|
2623
|
+
// session. A provided list with everything committed (resume) skips
|
|
2624
|
+
// coverage entirely.
|
|
2625
|
+
if (!requirementIdsProvided || pending.length > 0) {
|
|
2626
|
+
if (!requirementIdsProvided) {
|
|
2627
|
+
queue.push({
|
|
2628
|
+
id: segment.id,
|
|
2629
|
+
toolNames: segment.toolNames,
|
|
2630
|
+
prompt: `${input.basePrompt}\n\n${segment.instruction}`,
|
|
2631
|
+
});
|
|
2632
|
+
continue;
|
|
2633
|
+
}
|
|
2634
|
+
for (let i = 0; i < pending.length; i += FRONTEND_PLAN_COVERAGE_BATCH_SIZE) {
|
|
2635
|
+
const slice = pending.slice(i, i + FRONTEND_PLAN_COVERAGE_BATCH_SIZE);
|
|
2636
|
+
queue.push({
|
|
2637
|
+
id: `coverage-batch-${i / FRONTEND_PLAN_COVERAGE_BATCH_SIZE + 1}`,
|
|
2638
|
+
toolNames: segment.toolNames,
|
|
2639
|
+
coverageSlice: slice,
|
|
2640
|
+
prompt: `${input.basePrompt}\n\n${segment.instruction}\n\nCOVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them.`,
|
|
2641
|
+
});
|
|
2642
|
+
}
|
|
2643
|
+
}
|
|
2644
|
+
}
|
|
2645
|
+
let last;
|
|
2646
|
+
let index = 0;
|
|
2647
|
+
while (index < queue.length && index < FRONTEND_PLAN_BATCH_MAX_SESSIONS) {
|
|
2648
|
+
const session = queue[index];
|
|
2649
|
+
const remaining = (session.coverageSlice ?? []).filter((id) => !input.committedRequirementIds?.().has(id));
|
|
2650
|
+
// Resume/earlier-batch commits may already cover this slice.
|
|
2651
|
+
if (session.coverageSlice && remaining.length === 0) {
|
|
2652
|
+
index += 1;
|
|
2653
|
+
continue;
|
|
2654
|
+
}
|
|
2655
|
+
let prompt = session.prompt;
|
|
2656
|
+
if (session.coverageSlice) {
|
|
2657
|
+
prompt = prompt.replace(/COVERAGE BATCH: process ONLY these requirements in this session: .*/, `COVERAGE BATCH: process ONLY these requirements in this session: ${remaining.join(", ")}. Other requirements are handled by separate sessions; do not record them.`);
|
|
2658
|
+
}
|
|
2659
|
+
const committedBefore = input.committedFactCount();
|
|
2660
|
+
const customTools = input.segmentCustomTools(session.toolNames);
|
|
2661
|
+
const result = await input.piStepFn({
|
|
2662
|
+
...input.sessionOptions,
|
|
2663
|
+
prompt,
|
|
2664
|
+
...(customTools.length > 0
|
|
2665
|
+
? {
|
|
2666
|
+
writerToolPolicy: {
|
|
2667
|
+
requireSdk: true,
|
|
2668
|
+
customTools,
|
|
2669
|
+
},
|
|
2670
|
+
}
|
|
2671
|
+
: {}),
|
|
2672
|
+
});
|
|
2673
|
+
last = result;
|
|
2674
|
+
try {
|
|
2675
|
+
await input.flushLedger();
|
|
2676
|
+
}
|
|
2677
|
+
catch {
|
|
2678
|
+
// best-effort: the node-level flush runs again after the attempt
|
|
2679
|
+
}
|
|
2680
|
+
const committedAfter = input.committedFactCount();
|
|
2681
|
+
if (result.ok) {
|
|
2682
|
+
index += 1;
|
|
2683
|
+
continue;
|
|
2684
|
+
}
|
|
2685
|
+
const committedFactsOnlySuccess = session.id !== "finalize" &&
|
|
2686
|
+
!(result.assistantText ?? "").trim() &&
|
|
2687
|
+
!result.stderr.trim() &&
|
|
2688
|
+
!result.timedOut &&
|
|
2689
|
+
committedAfter > committedBefore;
|
|
2690
|
+
if (committedFactsOnlySuccess) {
|
|
2691
|
+
// The session died but banked facts: keep the progress and move on.
|
|
2692
|
+
index += 1;
|
|
2693
|
+
continue;
|
|
2694
|
+
}
|
|
2695
|
+
// Option 5: a multi-requirement coverage batch that failed with ZERO
|
|
2696
|
+
// new facts and no provider stderr is the upfront-reasoning burn —
|
|
2697
|
+
// halve the slice and retry instead of failing the attempt.
|
|
2698
|
+
const coverageSlice = session.coverageSlice;
|
|
2699
|
+
const zeroProgressBurn = coverageSlice !== undefined &&
|
|
2700
|
+
coverageSlice.length > 1 &&
|
|
2701
|
+
committedAfter === committedBefore &&
|
|
2702
|
+
!(result.assistantText ?? "").trim() &&
|
|
2703
|
+
!result.stderr.trim() &&
|
|
2704
|
+
!result.timedOut;
|
|
2705
|
+
if (zeroProgressBurn && coverageSlice) {
|
|
2706
|
+
const half = Math.ceil(coverageSlice.length / 2);
|
|
2707
|
+
queue.splice(index, 1, { ...session, coverageSlice: coverageSlice.slice(0, half) }, { ...session, coverageSlice: coverageSlice.slice(half) });
|
|
2708
|
+
continue;
|
|
2709
|
+
}
|
|
2710
|
+
return result;
|
|
2711
|
+
}
|
|
2712
|
+
return (last ?? {
|
|
2713
|
+
ok: false,
|
|
2714
|
+
stdout: "",
|
|
2715
|
+
stderr: "frontend plan segmentation produced no session",
|
|
2716
|
+
failureCategory: "empty-output",
|
|
2717
|
+
durationMs: 0,
|
|
2718
|
+
});
|
|
2719
|
+
}
|
|
2527
2720
|
export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
|
|
2528
2721
|
const started = Date.now();
|
|
2529
2722
|
const persona = resolveDagPiPersona(input.task);
|
|
@@ -2906,10 +3099,9 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
2906
3099
|
const piExtensionPaths = piExtensionsResolution && resolvePiBackend() !== "cli-only"
|
|
2907
3100
|
? piExtensionsResolution.resolved.flatMap((entry) => entry.entryPaths)
|
|
2908
3101
|
: undefined;
|
|
2909
|
-
|
|
3102
|
+
const piSessionOptions = {
|
|
2910
3103
|
attachedFiles: [],
|
|
2911
3104
|
modelConfig: resolveDagPiModelConfig(input.model, input.thinking ? { thinking: input.thinking } : undefined),
|
|
2912
|
-
prompt: input.prompt,
|
|
2913
3105
|
repoRoot: input.cwd,
|
|
2914
3106
|
step,
|
|
2915
3107
|
toolNames: resolveDagPiToolNames(input.task),
|
|
@@ -2922,7 +3114,6 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
2922
3114
|
...(input.task.contextBudget
|
|
2923
3115
|
? { contextBudget: input.task.contextBudget }
|
|
2924
3116
|
: {}),
|
|
2925
|
-
...(writerToolPolicy ? { writerToolPolicy } : {}),
|
|
2926
3117
|
...(piExtensionPaths && piExtensionPaths.length > 0
|
|
2927
3118
|
? { piExtensionPaths }
|
|
2928
3119
|
: {}),
|
|
@@ -2935,7 +3126,55 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
2935
3126
|
bridgeActivity(activity.kind, activity.at);
|
|
2936
3127
|
}
|
|
2937
3128
|
: undefined,
|
|
2938
|
-
}
|
|
3129
|
+
};
|
|
3130
|
+
if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
|
|
3131
|
+
// Frontend-only split: three sequential sessions with independent
|
|
3132
|
+
// output budgets (coverage -> UX decisions -> finalize), mirroring the
|
|
3133
|
+
// backend-test template's module sharding. Every other template keeps
|
|
3134
|
+
// the single-session path below.
|
|
3135
|
+
let planRequirementIds = [];
|
|
3136
|
+
try {
|
|
3137
|
+
const { readCommittedOriginFacts } = await import("../workflows/dag/frontend-shadow-dual-write.js");
|
|
3138
|
+
const contractFacts = await readCommittedOriginFacts(meta.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl");
|
|
3139
|
+
// Only requirement facts: the contract ledger also carries
|
|
3140
|
+
// constraints (CON-*), evidence expectations (EV-*), handoff
|
|
3141
|
+
// intents (HND-*), open questions (OQ-*) and split proposals
|
|
3142
|
+
// (SPLIT-*) that all have ids — feeding those into the coverage
|
|
3143
|
+
// batches made the model record non-frozen plan-requirement ids
|
|
3144
|
+
// that finalize's canonical-coverage gate then rejected (r-ext2).
|
|
3145
|
+
planRequirementIds = contractFacts
|
|
3146
|
+
.filter((record) => record.fact
|
|
3147
|
+
?.kind === "requirement")
|
|
3148
|
+
.map((record) => record.fact?.id)
|
|
3149
|
+
.filter((id) => typeof id === "string")
|
|
3150
|
+
.sort();
|
|
3151
|
+
}
|
|
3152
|
+
catch {
|
|
3153
|
+
// Unreadable ledger falls back to a single coverage session.
|
|
3154
|
+
}
|
|
3155
|
+
result = await runFrontendPlanSegmentedSessions({
|
|
3156
|
+
piStepFn,
|
|
3157
|
+
sessionOptions: piSessionOptions,
|
|
3158
|
+
basePrompt: input.prompt,
|
|
3159
|
+
attempt: input.attempt ?? 1,
|
|
3160
|
+
committedFactCount: () => planLedgerTools.committedFactCount(),
|
|
3161
|
+
requirementIds: planRequirementIds,
|
|
3162
|
+
committedRequirementIds: () => planLedgerTools.committedRequirementIds(),
|
|
3163
|
+
segmentCustomTools: (toolNames) => toolNames === null
|
|
3164
|
+
? planLedgerTools.customTools
|
|
3165
|
+
: planLedgerTools.customTools.filter((tool) => typeof tool === "object" &&
|
|
3166
|
+
tool !== null &&
|
|
3167
|
+
toolNames.has(tool.name)),
|
|
3168
|
+
flushLedger: () => planLedgerTools.flush(),
|
|
3169
|
+
});
|
|
3170
|
+
}
|
|
3171
|
+
else {
|
|
3172
|
+
result = await piStepFn({
|
|
3173
|
+
...piSessionOptions,
|
|
3174
|
+
prompt: input.prompt,
|
|
3175
|
+
...(writerToolPolicy ? { writerToolPolicy } : {}),
|
|
3176
|
+
});
|
|
3177
|
+
}
|
|
2939
3178
|
}
|
|
2940
3179
|
catch (error) {
|
|
2941
3180
|
if (playwrightToolContext) {
|
|
@@ -3072,25 +3311,10 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
3072
3311
|
catch {
|
|
3073
3312
|
// best-effort flush; missing ledger still fails at the node validator
|
|
3074
3313
|
}
|
|
3075
|
-
//
|
|
3076
|
-
//
|
|
3077
|
-
//
|
|
3078
|
-
//
|
|
3079
|
-
// session log and retry with a reduced-reading instruction.
|
|
3080
|
-
const readBurstIssues = input.task.readBudget?.onExhaustion === "return-guidance"
|
|
3081
|
-
? []
|
|
3082
|
-
: await detectPlanReadBurst({
|
|
3083
|
-
runDir: meta.runDir,
|
|
3084
|
-
nodeId: input.task.id,
|
|
3085
|
-
});
|
|
3086
|
-
if (readBurstIssues.length > 0) {
|
|
3087
|
-
return {
|
|
3088
|
-
...mapped,
|
|
3089
|
-
ok: false,
|
|
3090
|
-
failureCategory: "read-burst",
|
|
3091
|
-
stderr: [mapped.stderr, ...readBurstIssues].filter(Boolean).join("\n\n"),
|
|
3092
|
-
};
|
|
3093
|
-
}
|
|
3314
|
+
// NOTE: the legacy plan read-burst guard lived here. It is dead code
|
|
3315
|
+
// since the plan node went tool-only (resolveDagPiToolNames grants no
|
|
3316
|
+
// read/grep/ls/find), so a read burst is structurally impossible; the
|
|
3317
|
+
// read-budget path (scout etc.) keeps its own guard.
|
|
3094
3318
|
}
|
|
3095
3319
|
if (!isWriteTask) {
|
|
3096
3320
|
if (!mapped.ok &&
|
|
@@ -90,6 +90,10 @@ export async function adoptTaskContract(input) {
|
|
|
90
90
|
requirementBytes: bytes.requirementBytes,
|
|
91
91
|
constraintsBytes: bytes.constraintsBytes,
|
|
92
92
|
referenceManifestBytes: bytes.referenceManifestBytes,
|
|
93
|
+
...(state.ref?.sourceHashes.requirementLedgerSha256 &&
|
|
94
|
+
bytes.requirementLedgerBytes !== null
|
|
95
|
+
? { requirementLedgerBytes: bytes.requirementLedgerBytes }
|
|
96
|
+
: {}),
|
|
93
97
|
});
|
|
94
98
|
const taskConfigSha256 = computeTaskConfigSha256(fullConfig);
|
|
95
99
|
const canonicalHash = computeCanonicalHash({
|
|
@@ -45,6 +45,10 @@ export async function refreshManagedContractAfterImport(input) {
|
|
|
45
45
|
requirementBytes: bytes.requirementBytes,
|
|
46
46
|
constraintsBytes: bytes.constraintsBytes,
|
|
47
47
|
referenceManifestBytes: bytes.referenceManifestBytes,
|
|
48
|
+
...(state2.ref.sourceHashes.requirementLedgerSha256 &&
|
|
49
|
+
bytes.requirementLedgerBytes !== null
|
|
50
|
+
? { requirementLedgerBytes: bytes.requirementLedgerBytes }
|
|
51
|
+
: {}),
|
|
48
52
|
});
|
|
49
53
|
const taskConfigSha256 = computeTaskConfigSha256(fullConfig);
|
|
50
54
|
const canonicalHash = computeCanonicalHash({
|
|
@@ -539,6 +539,18 @@ export const frontendImplementationContractSchema = z
|
|
|
539
539
|
message: `unknown verification target ${targetId}`,
|
|
540
540
|
path: ["requirements"],
|
|
541
541
|
});
|
|
542
|
+
// Interactions bind to VTs too: an interaction whose behavior
|
|
543
|
+
// verification points at an unmaterialized VT is untraceable —
|
|
544
|
+
// r20 review finding (dangling *-BEHAVIOR references sailed
|
|
545
|
+
// through plan/design/implement to the final review).
|
|
546
|
+
for (const interaction of value.interactions)
|
|
547
|
+
for (const targetId of interaction.verificationTargetIds)
|
|
548
|
+
if (!verificationIds.includes(targetId))
|
|
549
|
+
ctx.addIssue({
|
|
550
|
+
code: "custom",
|
|
551
|
+
message: `interaction "${interaction.name}" references unknown verification target ${targetId}`,
|
|
552
|
+
path: ["interactions"],
|
|
553
|
+
});
|
|
542
554
|
// Source fidelity ledger binding (AC-005/AC-006): 绑定携带
|
|
543
555
|
// requirementToFragments(ledger v2)时,每个 requirement 必须携带
|
|
544
556
|
// 非空 sourceFragmentIds,否则 fail closed(防引用伪造/缺失)。
|
|
@@ -570,12 +582,23 @@ export const frontendImplementationContractSchema = z
|
|
|
570
582
|
if (state.applicable &&
|
|
571
583
|
(!state.expectedBehavior ||
|
|
572
584
|
!state.implementationTargets?.length ||
|
|
573
|
-
!state.verificationTargetIds?.length))
|
|
585
|
+
!state.verificationTargetIds?.length)) {
|
|
586
|
+
// Name the state and the exact missing fields: the fixer is a
|
|
587
|
+
// model iterating on finalize receipts — it cannot fix a
|
|
588
|
+
// defect it cannot locate (r18: 39 blind finalize retries).
|
|
589
|
+
const missing = [];
|
|
590
|
+
if (!state.expectedBehavior)
|
|
591
|
+
missing.push("expectedBehavior");
|
|
592
|
+
if (!state.implementationTargets?.length)
|
|
593
|
+
missing.push("implementationTargets");
|
|
594
|
+
if (!state.verificationTargetIds?.length)
|
|
595
|
+
missing.push("verificationTargetIds");
|
|
574
596
|
ctx.addIssue({
|
|
575
597
|
code: "custom",
|
|
576
|
-
message:
|
|
598
|
+
message: `applicable UI state "${state.name}" is missing: ${missing.join(", ")} — record_state_flow it again with those fields filled`,
|
|
577
599
|
path: ["uiStates"],
|
|
578
600
|
});
|
|
601
|
+
}
|
|
579
602
|
if (!state.applicable && !state.notApplicableReason)
|
|
580
603
|
ctx.addIssue({
|
|
581
604
|
code: "custom",
|
|
@@ -1826,12 +1849,25 @@ export async function analyzeFrontendImplementationContract(input) {
|
|
|
1826
1849
|
if (parsedTargetFiles.some((file) => file.startsWith("/") || file.includes("\\")))
|
|
1827
1850
|
fail("blocked", "invalid-output: frontend contract target paths must be relative POSIX paths", candidateJsonSha256);
|
|
1828
1851
|
const parsedStates = asRecord(parsed)?.uiStates;
|
|
1829
|
-
if (Array.isArray(parsedStates)
|
|
1830
|
-
const
|
|
1831
|
-
|
|
1832
|
-
|
|
1833
|
-
|
|
1834
|
-
|
|
1852
|
+
if (Array.isArray(parsedStates)) {
|
|
1853
|
+
const incompleteStates = [];
|
|
1854
|
+
for (const item of parsedStates) {
|
|
1855
|
+
const state = asRecord(item);
|
|
1856
|
+
if (!state || state.applicable !== true)
|
|
1857
|
+
continue;
|
|
1858
|
+
const missing = [];
|
|
1859
|
+
if (!asString(state.expectedBehavior))
|
|
1860
|
+
missing.push("expectedBehavior");
|
|
1861
|
+
if (asStringArray(state.implementationTargets).length === 0)
|
|
1862
|
+
missing.push("implementationTargets");
|
|
1863
|
+
if (asStringArray(state.verificationTargetIds).length === 0)
|
|
1864
|
+
missing.push("verificationTargetIds");
|
|
1865
|
+
if (missing.length > 0)
|
|
1866
|
+
incompleteStates.push(`"${asString(state.name)}": missing ${missing.join(", ")}`);
|
|
1867
|
+
}
|
|
1868
|
+
if (incompleteStates.length > 0)
|
|
1869
|
+
fail("retryable-invalid", `invalid-output: applicable UI states incomplete — re-record each with record_state_flow filling the named fields: ${incompleteStates.join("; ")}`, candidateJsonSha256);
|
|
1870
|
+
}
|
|
1835
1871
|
const parsedMockApi = asRecord(parsed)?.mockApi;
|
|
1836
1872
|
if (asRecord(parsedMockApi) &&
|
|
1837
1873
|
typeof asRecord(parsedMockApi)?.strategy === "string" &&
|
|
@@ -2079,7 +2115,7 @@ const VERIFICATION_SYMBOL_MAX_CHARS = 60;
|
|
|
2079
2115
|
* loading") and `describe(...)` / `it(...)` forms stay valid — a real symbol
|
|
2080
2116
|
* may be a function name, a dotted path, or a describe/it title.
|
|
2081
2117
|
*/
|
|
2082
|
-
function isSuspiciousVerificationSymbol(symbol) {
|
|
2118
|
+
export function isSuspiciousVerificationSymbol(symbol) {
|
|
2083
2119
|
const trimmed = symbol.trim();
|
|
2084
2120
|
if (!trimmed)
|
|
2085
2121
|
return false;
|
|
@@ -2126,6 +2162,17 @@ export class PlanPolicyPrecheckFailure extends Error {
|
|
|
2126
2162
|
*/
|
|
2127
2163
|
export async function analyzeFrontendPlanPatchCandidate(input) {
|
|
2128
2164
|
const analysis = await analyzeFrontendImplementationContract(input);
|
|
2165
|
+
// A committed VT with a fabricated symbol is an immutable ledger fact —
|
|
2166
|
+
// the record boundary rejects duplicate ids, so the model cannot overwrite
|
|
2167
|
+
// it and throwing here deadlocks the receipt loop (r19: VT-AC006-BEHAVIOR).
|
|
2168
|
+
// Drop suspicious symbols deterministically instead: the VT stays valid and
|
|
2169
|
+
// the deterministic trace gate verifies file+command (and resolvability)
|
|
2170
|
+
// after verification. Mirrors the shell materialization's drop semantics.
|
|
2171
|
+
for (const target of analysis.canonical.verificationTargets) {
|
|
2172
|
+
if (target.symbol && isSuspiciousVerificationSymbol(target.symbol)) {
|
|
2173
|
+
target.symbol = undefined;
|
|
2174
|
+
}
|
|
2175
|
+
}
|
|
2129
2176
|
// Front-load the verification-symbol shape check so fabricated symbols
|
|
2130
2177
|
// are fixed by the plan retry ladder in-node instead of failing the
|
|
2131
2178
|
// verify trace gate at the end of the run.
|