@tea-agent/loop-agent 0.42.0-next.8 → 0.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +56 -50
- package/dist/application/dag/run-dag.js +41 -0
- package/dist/application/task-lifecycle/advance.js +9 -4
- package/dist/build-stamp.json +3 -3
- package/dist/cli/program.js +1 -1
- package/dist/commands/dag-rerun-task.js +2 -0
- package/dist/commands/task-source-prepare.js +3 -1
- package/dist/executors/dag-pi-executor.js +1809 -213
- package/dist/executors/pi-extension-resolver.js +14 -2
- package/dist/executors/shell-executor.js +178 -73
- package/dist/shared/dag-failure-category.js +6 -0
- package/dist/task/contract/apply.js +36 -2
- package/dist/task/source-prepare/parse-intent.js +7 -0
- package/dist/worker/console/chat/chat-event-store.js +4 -2
- package/dist/worker/console/chat/pi-runtime.js +45 -2
- package/dist/worker/console/chat/resource-loader.js +4 -1
- package/dist/worker/console/chat/routes.js +28 -12
- package/dist/worker/console/chat/sdd-data-alignment.js +13 -2
- package/dist/worker/console/chat/session-store.js +5 -1
- package/dist/worker/console/chat/subagents/agent-tool.js +62 -0
- package/dist/worker/console/chat/subagents/explore-agent.js +169 -0
- package/dist/worker/console/chat/subagents/index.js +5 -0
- package/dist/worker/console/chat/subagents/orchestrator.js +273 -0
- package/dist/worker/console/chat/subagents/tool-policy.js +53 -0
- package/dist/worker/console/chat/subagents/types.js +13 -0
- package/dist/worker/console/chat/tool-preview.js +10 -0
- package/dist/worker/console/chat/tools.js +11 -1
- package/dist/worker/console/chat/turn-process.js +1 -0
- package/dist/worker/console/interview/tools.js +1 -0
- package/dist/worker/console/static/assets/{abnfDiagram-N423BO3Z-C6n9_m0P.js → abnfDiagram-N423BO3Z-B0Q0nClf.js} +1 -1
- package/dist/worker/console/static/assets/{arc-DQh-IfZ1.js → arc-DCPjC19G.js} +1 -1
- package/dist/worker/console/static/assets/{architectureDiagram-T3A2C74G-54NnrwUC.js → architectureDiagram-T3A2C74G-CP5n9jmG.js} +1 -1
- package/dist/worker/console/static/assets/{blockDiagram-VBNYF7ZC-pivRALGK.js → blockDiagram-VBNYF7ZC-CmaWi_Wg.js} +1 -1
- package/dist/worker/console/static/assets/{c4Diagram-5PPSVZJV-BR7OV2NJ.js → c4Diagram-5PPSVZJV-Ct5tcMfZ.js} +1 -1
- package/dist/worker/console/static/assets/channel-DAS07MdS.js +1 -0
- package/dist/worker/console/static/assets/{chunk-2GRJ4B5K-B1Aq6BcK.js → chunk-2GRJ4B5K-C0osm1Zf.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-2Q5K7J3B-CbdU0rjo.js → chunk-2Q5K7J3B-CDWviED7.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-5RXB4S5H-Dqi2GJeD.js → chunk-5RXB4S5H-mtSk1-j7.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-5VM5RSS4-BYn1Hu4R.js → chunk-5VM5RSS4-D5J9PHC7.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-6Q2QTUOP-DRLjDL0k.js → chunk-6Q2QTUOP-Begg4WAa.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-GF5L2VYU-CY9Xx2jV.js → chunk-GF5L2VYU-sF1AEy7T.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-JWPE2WC7-BNCZs7_z.js → chunk-JWPE2WC7-BqbsWXZb.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-KBJHAD2P-DZ8AStLL.js → chunk-KBJHAD2P-BKJrPcJ6.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-RYQCIY6F-DsxjYrzz.js → chunk-RYQCIY6F-CcwNMRho.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-XXDRQBXY-Ddx1KC1l.js → chunk-XXDRQBXY-BEJi3iIs.js} +1 -1
- package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-Cwk-5jiW.js +1 -0
- package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-Cwk-5jiW.js +1 -0
- package/dist/worker/console/static/assets/{cose-bilkent-JH36ORCC-W9TveCnK.js → cose-bilkent-JH36ORCC-DRLgTvVu.js} +1 -1
- package/dist/worker/console/static/assets/{cynefin-VYW2F7L2-l9j_ztZH.js → cynefin-VYW2F7L2-BoAVG3cs.js} +1 -1
- package/dist/worker/console/static/assets/{cynefinDiagram-MW4NZA55-BMJMKi4G.js → cynefinDiagram-MW4NZA55-Bzol06hR.js} +1 -1
- package/dist/worker/console/static/assets/{dagre-VZM6K2ZE-Bb4mG9pH.js → dagre-VZM6K2ZE-UmliJM77.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-7IWD3JNH-BMgL_Qv0.js → diagram-7IWD3JNH-BWLcJkFo.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-B4RE2ZJO-Bvo2T4OQ.js → diagram-B4RE2ZJO-CgApOm-O.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-LBJQPF4R-_5kWRN9c.js → diagram-LBJQPF4R-Bh3RgTLs.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-Q27KOJAE-DqdMrltM.js → diagram-Q27KOJAE-Dsh-D5nE.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-UB23O5K3-CDDYsNkp.js → diagram-UB23O5K3-Dkbbpcpb.js} +1 -1
- package/dist/worker/console/static/assets/{ebnfDiagram-BXEA7PRR-c4nfnZUV.js → ebnfDiagram-BXEA7PRR-3EBaA3t3.js} +1 -1
- package/dist/worker/console/static/assets/{erDiagram-JOGREHBK-DnkUcNMs.js → erDiagram-JOGREHBK-Bq2Dc5ok.js} +1 -1
- package/dist/worker/console/static/assets/{flowDiagram-UKHOOZJN-Dz0KGLZE.js → flowDiagram-UKHOOZJN-De7y55ha.js} +1 -1
- package/dist/worker/console/static/assets/{ganttDiagram-PKOTCBZU-Cj1t1uka.js → ganttDiagram-PKOTCBZU-fLiDbQRh.js} +1 -1
- package/dist/worker/console/static/assets/{gitGraphDiagram-DS77QQ5N-tp2FrBHd.js → gitGraphDiagram-DS77QQ5N-Du617mp1.js} +1 -1
- package/dist/worker/console/static/assets/index-C0O48S_P.js +449 -0
- package/dist/worker/console/static/assets/index-CzKf4U8P.css +1 -0
- package/dist/worker/console/static/assets/{infoDiagram-6WML65LV-DQwJS8DH.js → infoDiagram-6WML65LV-BwmlF-XP.js} +1 -1
- package/dist/worker/console/static/assets/{ishikawaDiagram-WSZJBQD7-eimCQWD4.js → ishikawaDiagram-WSZJBQD7-zsGtS59u.js} +1 -1
- package/dist/worker/console/static/assets/{journeyDiagram-NVQOT4AX-Dd9Gbwgv.js → journeyDiagram-NVQOT4AX-DTTPTe3f.js} +1 -1
- package/dist/worker/console/static/assets/{kanban-definition-27J2QSJJ-Qve3cyft.js → kanban-definition-27J2QSJJ-BkXg9aP-.js} +1 -1
- package/dist/worker/console/static/assets/{linear-UXKSe36Z.js → linear-7U2ue5IE.js} +1 -1
- package/dist/worker/console/static/assets/{mermaid.core-CWhj4JXN.js → mermaid.core-BUuGHmWO.js} +5 -5
- package/dist/worker/console/static/assets/{mindmap-definition-FAOFIHXS-yTcwf4SJ.js → mindmap-definition-FAOFIHXS-7q6fX0Sv.js} +1 -1
- package/dist/worker/console/static/assets/{pegDiagram-VL7TDLO6-_7qToaZP.js → pegDiagram-VL7TDLO6-CkQPJN55.js} +1 -1
- package/dist/worker/console/static/assets/{pieDiagram-7S7Q4E2Y-CXmrKmDm.js → pieDiagram-7S7Q4E2Y-CCal_tfX.js} +1 -1
- package/dist/worker/console/static/assets/{quadrantDiagram-CIZ2JOQS-hMmrJyaS.js → quadrantDiagram-CIZ2JOQS-B7bpxyAZ.js} +1 -1
- package/dist/worker/console/static/assets/{railroadDiagram-AXF67PYL-BXw8tCBe.js → railroadDiagram-AXF67PYL-CJmww7D-.js} +1 -1
- package/dist/worker/console/static/assets/{requirementDiagram-LRYGKXZP-BmjgnLZl.js → requirementDiagram-LRYGKXZP-D6ldWJEv.js} +1 -1
- package/dist/worker/console/static/assets/{sankeyDiagram-W5VNT64P-Bj77SPpN.js → sankeyDiagram-W5VNT64P-VGh33I9e.js} +1 -1
- package/dist/worker/console/static/assets/{sequenceDiagram-SI44F4Z6-B2FDVysL.js → sequenceDiagram-SI44F4Z6-CCkDrIjU.js} +1 -1
- package/dist/worker/console/static/assets/{sizeCapture-X5ZJPWSS-UChNY9ZZ.js → sizeCapture-X5ZJPWSS-NC8f5Otb.js} +1 -1
- package/dist/worker/console/static/assets/{stateDiagram-OKZ733FA-wbAEkbd7.js → stateDiagram-OKZ733FA-_92ezZdF.js} +1 -1
- package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-DyJg5gzW.js +1 -0
- package/dist/worker/console/static/assets/{swimlanes-SLNWSIFB-DCEkU8HR.js → swimlanes-SLNWSIFB-BdUCGtbP.js} +2 -2
- package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-B1cw_0Uy.js +8 -0
- package/dist/worker/console/static/assets/{timeline-definition-Z64GVDOM-BGp7H06U.js → timeline-definition-Z64GVDOM-CdD1H8ia.js} +1 -1
- package/dist/worker/console/static/assets/{vennDiagram-T6HMQDX7-DsCIQuWR.js → vennDiagram-T6HMQDX7-Dkdk3oFo.js} +1 -1
- package/dist/worker/console/static/assets/{wardleyDiagram-T6FBY63Y-cRZLP4Xe.js → wardleyDiagram-T6FBY63Y-CKu2uPPY.js} +1 -1
- package/dist/worker/console/static/assets/{xychartDiagram-ELKLHX3M-CP0xLR9Y.js → xychartDiagram-ELKLHX3M-ClqAnuG8.js} +1 -1
- package/dist/worker/console/static/index.html +2 -2
- package/dist/worker/console/static-src/chat-view-types.js +1 -1
- package/dist/worker/console/static-src/operator-chat/chat-sse-events.js +23 -0
- package/dist/worker/console/static-src/operator-chat/pending-user-message.js +32 -0
- package/dist/worker/console/static-src/operator-chat/tools-catalog.js +21 -2
- package/dist/worker/console/static-src/operator-chat/turn-stream-controller.js +2 -0
- package/dist/worker/console/static-src/operator-chat/useChatSessions.js +2 -0
- package/dist/worker/console/static-src/operator-chat/useChatStream.js +30 -14
- package/dist/worker/console/static-src/operator-chat/useChatThread.js +133 -5
- package/dist/worker/console/static-src/shell/workspace-route.js +4 -0
- package/dist/worker/observe/routes.js +4 -0
- package/dist/worker/observe/static/constants.js +22 -22
- package/dist/worker/observe/static/dag-context-reason-labels.js +19 -0
- package/dist/worker/observe/static/dag-history-labels.js +1 -0
- package/dist/worker/observe/static/dag-inspector-humanize.d.ts +16 -0
- package/dist/worker/observe/static/dag-inspector-humanize.js +339 -0
- package/dist/worker/observe/static/dag-node-purpose.js +10 -10
- package/dist/worker/observe/static/index.html +4 -4
- package/dist/worker/observe/static/prompt-restart-candidates.js +4 -2
- package/dist/worker/observe/static/relations.js +5 -5
- package/dist/worker/observe/static/styles.css +171 -0
- package/dist/worker/observe/static/views/dag-graph.js +1 -1
- package/dist/worker/observe/static/views/dag-inspector.js +329 -174
- package/dist/worker/observe/static/views/dag.js +16 -12
- package/dist/worker/observe/static/views/session-timeline.js +5 -3
- package/dist/workflows/dag/backend-test-case-coverage-analysis.js +18 -0
- package/dist/workflows/dag/backend-test-plan-protocol.js +104 -0
- package/dist/workflows/dag/backend-test-scenario-param.js +30 -1
- package/dist/workflows/dag/dag-retry-schema.js +3 -0
- package/dist/workflows/dag/frontend-contract-facts.js +130 -0
- package/dist/workflows/dag/frontend-design-policy.js +101 -16
- package/dist/workflows/dag/frontend-implementation-contract.js +410 -41
- package/dist/workflows/dag/frontend-plan-render.js +13 -2
- package/dist/workflows/dag/frontend-prewrite-gate.js +1 -1
- package/dist/workflows/dag/frontend-recovery-controller.js +25 -1
- package/dist/workflows/dag/frontend-recovery-plan.js +6 -2
- package/dist/workflows/dag/frontend-recovery-run.js +169 -15
- package/dist/workflows/dag/frontend-risk.js +2 -0
- package/dist/workflows/dag/frontend-shadow-dual-write.js +17 -1
- package/dist/workflows/dag/frontend-shape.js +16 -6
- package/dist/workflows/dag/frontend-test-execution-evidence.js +48 -0
- package/dist/workflows/dag/frontend-typed-event-store.js +3 -0
- package/dist/workflows/dag/frontend-verification-trace.js +38 -4
- package/dist/workflows/dag/frontend-writer-admission.js +46 -12
- package/dist/workflows/dag/init-hybrid.js +59 -29
- package/dist/workflows/dag/node-execution.js +277 -85
- package/dist/workflows/dag/recovery-lease.js +106 -16
- package/dist/workflows/dag/rerun-feedback.js +60 -0
- package/dist/workflows/dag/rerun-plan.js +90 -4
- package/dist/workflows/dag/rerun-run.js +28 -6
- package/dist/workflows/dag/rerun-task.js +224 -15
- package/dist/workflows/dag/retry-policy.js +27 -10
- package/dist/workflows/dag/runner-exit-diagnostics.js +125 -0
- package/dist/workflows/dag/runner.js +238 -29
- package/dist/workflows/dag/scheduler.js +21 -6
- package/dist/workflows/dag/structured-output-repair.js +4 -1
- package/dist/workflows/dag/types.js +4 -3
- package/dist/workflows/dag/validate.js +10 -8
- package/dist/workflows/dag/workspace-checkpoint.js +66 -0
- package/docs/operations/README.md +1 -0
- package/docs/templates/agent-dag.schema.json +2 -2
- package/docs/templates/backend-test-dag.json +6 -4
- package/docs/templates/frontend-implementation-contract.schema.json +34 -2
- package/harness.json +1 -1
- package/package.json +4 -5
- package/skills/frontend-plan/SKILL.md +14 -1
- package/skills/frontend-plan/references/decision-contract.md +101 -5
- package/skills/frontend-plan/references/design-decisions.md +32 -0
- package/dist/worker/console/static/assets/channel-DsZxrgqe.js +0 -1
- package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-CuToPeGV.js +0 -1
- package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-CuToPeGV.js +0 -1
- package/dist/worker/console/static/assets/index-D9Sc0f0y.js +0 -449
- package/dist/worker/console/static/assets/index-DDQc5a50.css +0 -1
- package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-yPoT19ft.js +0 -1
- package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-VWdarchG.js +0 -8
|
@@ -12,25 +12,255 @@ import { createPiReadBudgetCustomTools, } from "./pi-read-budget-policy.js";
|
|
|
12
12
|
import { cleanupPlaywrightCliDefaultSession, createPlaywrightCliTool, PI_COMMAND_CAPABILITY_REGISTRY, resolveCaseIdFromWriteSet, resolveEvidenceDirFromWriteSet, } from "./pi-playwright-cli-tool.js";
|
|
13
13
|
import { dagCommandPolicyAllows, resolveDagCommandPolicy, } from "../workflows/dag/types.js";
|
|
14
14
|
import { parseLedgerJson } from "../task/source-prepare/ledger.js";
|
|
15
|
+
import { splitDocumentIntoFragments, sha256Text, } from "../task/source-prepare/fragment-inventory.js";
|
|
15
16
|
import { redactPromptForLog, truncateOutput, } from "../shared/output-truncation.js";
|
|
16
17
|
import { GitStatusUnavailableError, pathsChangedDuringRun, readGitStatusPorcelain, recoverRootNulArtifact, snapshotGitStatusPathFingerprints, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
|
|
17
18
|
import { captureWorkspaceWriteSnapshot, diffWorkspaceWriteSnapshots, } from "./workspace-write-snapshot.js";
|
|
18
19
|
import { pathMatchesPattern } from "../shared/git-progress.js";
|
|
19
|
-
import {
|
|
20
|
+
import { readTypedEventStoreFromJsonl, typedEventPayloadSha256, } from "../workflows/dag/frontend-typed-event-store.js";
|
|
21
|
+
import { collectCanonicalStateFlowNames, frontendEvidenceExpectationSchema, resolveFrontendContractRequirements, validateFrontendRequiredDeliverables, } from "../workflows/dag/frontend-contract-facts.js";
|
|
22
|
+
import { isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, OUTPUT_LIMIT_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "../workflows/dag/retry-policy.js";
|
|
20
23
|
import { assessBackendTestMdPlanCompleteness, assessBackendTestMdWriterCompleteness, assessBackendTestPytestPlanCompleteness, assessBackendTestPytestWriterCompleteness, assessBackendTestShardChildCompleteness, backendTestWriterProgressRoleForTask, classifyBackendTestWriterCompletenessFailure, isBackendTestCompletenessRetryCandidate, isBackendTestMdPlanTask, isBackendTestPytestCollectionRepairOutcomeRecoveryCandidate, isBackendTestPytestPlanTask, isBackendTestShardChildTask, writeBackendTestWriterProgressArtifacts, } from "../workflows/dag/backend-test-writer-completeness.js";
|
|
21
24
|
import { resolveBackendTestLayout } from "../workflows/dag/backend-test-layout.js";
|
|
25
|
+
import { assessBackendTestPlanProtocol } from "../workflows/dag/backend-test-plan-protocol.js";
|
|
22
26
|
import { frontendTestLayoutFromSpec } from "../workflows/dag/frontend-test-layout.js";
|
|
23
27
|
import { redactSecrets, truncateUtf8Preview } from "../shared/preview.js";
|
|
24
28
|
import { writeEffectiveContextReceipt } from "../workflows/dag/context-receipt.js";
|
|
25
29
|
/**
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
* and produced zero attributed diff. This is a terminal, non-retryable
|
|
29
|
-
* diagnosis (the same prompt + model will hit the same budget wall); the
|
|
30
|
-
* recommendation is a model switch plus a fresh run. It must NOT mask a
|
|
31
|
-
* recoverable partial-write-set (incomplete-write-set) upgrade.
|
|
30
|
+
* Legacy diagnostic label retained for artifact compatibility. New executions
|
|
31
|
+
* classify this signal as output-limit so the node can retry incrementally.
|
|
32
32
|
*/
|
|
33
33
|
export const WRITER_THINKING_EXHAUSTED_CATEGORY = "writer-thinking-exhausted";
|
|
34
|
+
/**
|
|
35
|
+
* Legacy diagnostic label retained for artifact compatibility. New executions
|
|
36
|
+
* classify this signal as output-limit and preserve committed typed facts.
|
|
37
|
+
*/
|
|
38
|
+
export const PLANNER_THINKING_EXHAUSTED_CATEGORY = "planner-thinking-exhausted";
|
|
39
|
+
/**
|
|
40
|
+
* Parallel coverage shards namespace their verification target ids with
|
|
41
|
+
* `VT-SHARD-<shard>-` so the reducer can detect cross-shard conflicts. Once
|
|
42
|
+
* every shard's facts are known the namespace must be restored: the canonical
|
|
43
|
+
* contract has to carry the exact ids the task source froze (a surviving
|
|
44
|
+
* `VT-SHARD-N-…` id is a contract-requirement gap that design review
|
|
45
|
+
* rejects). Stripping happens per shard before adoption — after the strip,
|
|
46
|
+
* the existing identity-conflict check catches genuine cross-shard
|
|
47
|
+
* duplicate base ids and fails the merge closed.
|
|
48
|
+
*/
|
|
49
|
+
/**
|
|
50
|
+
* Deterministic pre-reduce for parallel coverage shard facts. Shards partition
|
|
51
|
+
* requirements, but a frozen verification target can legitimately be derived
|
|
52
|
+
* by several shards (one VT covers multiple ACs across shard boundaries), so
|
|
53
|
+
* same-id verification targets are MERGED: requirementIds and uiStates union,
|
|
54
|
+
* while divergent file/commandId is a real conflict. Everything else passes
|
|
55
|
+
* through unchanged.
|
|
56
|
+
*/
|
|
57
|
+
export function reduceParallelCoverageShardRecords(records) {
|
|
58
|
+
// Resolve replacements inside each source shard before comparing shards.
|
|
59
|
+
// A correction is an event-log operation, not a second independent target;
|
|
60
|
+
// retaining both records makes a valid same-shard correction look like a
|
|
61
|
+
// cross-shard conflict during the later reduction.
|
|
62
|
+
const byShard = new Map();
|
|
63
|
+
for (const record of records) {
|
|
64
|
+
const shard = byShard.get(record.attemptId) ?? [];
|
|
65
|
+
shard.push(record);
|
|
66
|
+
byShard.set(record.attemptId, shard);
|
|
67
|
+
}
|
|
68
|
+
const effectiveRecords = [];
|
|
69
|
+
for (const shardRecords of byShard.values()) {
|
|
70
|
+
const effective = [];
|
|
71
|
+
for (const record of shardRecords) {
|
|
72
|
+
const fact = record.fact;
|
|
73
|
+
const entry = fact?.entry;
|
|
74
|
+
const kind = typeof fact?.kind === "string" ? fact.kind : "";
|
|
75
|
+
const identity = (kind === "plan-requirement" || kind === "plan-verification-target") &&
|
|
76
|
+
typeof entry?.id === "string"
|
|
77
|
+
? `${kind}:${entry.id}`
|
|
78
|
+
: undefined;
|
|
79
|
+
if (!identity) {
|
|
80
|
+
effective.push(record);
|
|
81
|
+
continue;
|
|
82
|
+
}
|
|
83
|
+
const replaces = typeof fact?.replaces === "string" ? `${kind}:${fact.replaces}` : undefined;
|
|
84
|
+
if (replaces) {
|
|
85
|
+
for (let index = effective.length - 1; index >= 0; index -= 1) {
|
|
86
|
+
const prior = effective[index];
|
|
87
|
+
const priorFact = prior.fact;
|
|
88
|
+
const priorEntry = priorFact?.entry;
|
|
89
|
+
const priorIdentity = typeof priorFact?.kind === "string" &&
|
|
90
|
+
typeof priorEntry?.id === "string"
|
|
91
|
+
? `${priorFact.kind}:${priorEntry.id}`
|
|
92
|
+
: undefined;
|
|
93
|
+
if (priorIdentity === replaces)
|
|
94
|
+
effective.splice(index, 1);
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
// Keep the source record immutable. The replacement is represented by
|
|
98
|
+
// the latest event and its own payload hash.
|
|
99
|
+
effective.push(record);
|
|
100
|
+
}
|
|
101
|
+
effectiveRecords.push(...effective);
|
|
102
|
+
}
|
|
103
|
+
const verificationTargets = new Map();
|
|
104
|
+
const merged = [];
|
|
105
|
+
for (const record of effectiveRecords) {
|
|
106
|
+
const fact = record.fact;
|
|
107
|
+
if (!fact || typeof fact !== "object")
|
|
108
|
+
continue;
|
|
109
|
+
if (fact.kind !== "plan-verification-target") {
|
|
110
|
+
merged.push(record);
|
|
111
|
+
continue;
|
|
112
|
+
}
|
|
113
|
+
const entry = fact.entry;
|
|
114
|
+
if (!entry || typeof entry.id !== "string")
|
|
115
|
+
continue;
|
|
116
|
+
const existing = verificationTargets.get(entry.id);
|
|
117
|
+
if (!existing) {
|
|
118
|
+
verificationTargets.set(entry.id, {
|
|
119
|
+
record, entry: { ...entry },
|
|
120
|
+
sources: [`${record.attemptId}:${record.eventId}:${record.payloadSha256}`],
|
|
121
|
+
});
|
|
122
|
+
continue;
|
|
123
|
+
}
|
|
124
|
+
// Never mutate the entry held by the source record. The reducer creates
|
|
125
|
+
// a new merged entry and recomputes the payload hash for that derived
|
|
126
|
+
// fact, leaving source-event replay integrity intact.
|
|
127
|
+
const baseEntry = { ...existing.entry };
|
|
128
|
+
for (const field of ["commandId", "commandLabel", "file", "scope"]) {
|
|
129
|
+
if (entry[field] !== baseEntry[field]) {
|
|
130
|
+
throw new Error(`frontend plan coverage shard reduce conflict: verification target ${entry.id} has divergent ${field} "${String(entry[field])}" vs "${String(baseEntry[field])}"`);
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
const unionSorted = (a, b) => {
|
|
134
|
+
const left = Array.isArray(a) ? a : [];
|
|
135
|
+
const right = Array.isArray(b) ? b : [];
|
|
136
|
+
return [...new Set([...left, ...right])].sort();
|
|
137
|
+
};
|
|
138
|
+
baseEntry.requirementIds = unionSorted(baseEntry.requirementIds, entry.requirementIds);
|
|
139
|
+
baseEntry.uiStates = unionSorted(baseEntry.uiStates, entry.uiStates);
|
|
140
|
+
const mergedFact = { ...fact, entry: { ...baseEntry } };
|
|
141
|
+
existing.entry = baseEntry;
|
|
142
|
+
existing.sources.push(`${record.attemptId}:${record.eventId}:${record.payloadSha256}`);
|
|
143
|
+
existing.record = {
|
|
144
|
+
...existing.record,
|
|
145
|
+
fact: mergedFact,
|
|
146
|
+
payloadSha256: typedEventPayloadSha256(mergedFact),
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
merged.push(...[...verificationTargets.values()].map((item) => {
|
|
150
|
+
// The reducer owns every VT revision, including a one-shard partial
|
|
151
|
+
// result. Content-addressed revisions replay idempotently, while an
|
|
152
|
+
// expanded/replaced source set appends an explicit replacement rather
|
|
153
|
+
// than masquerading as a changed source event or deleting audit facts.
|
|
154
|
+
const fact = {
|
|
155
|
+
...item.record.fact,
|
|
156
|
+
replaces: item.entry.id,
|
|
157
|
+
entry: {
|
|
158
|
+
...item.entry,
|
|
159
|
+
requirementIds: [...new Set(planFactStringList(item.entry.requirementIds))].sort(),
|
|
160
|
+
uiStates: [...new Set(planFactStringList(item.entry.uiStates))].sort(),
|
|
161
|
+
},
|
|
162
|
+
};
|
|
163
|
+
const payloadSha256 = typedEventPayloadSha256(fact);
|
|
164
|
+
const revisionId = createHash("sha256")
|
|
165
|
+
.update(JSON.stringify({ sources: [...new Set(item.sources)].sort(), payloadSha256 }))
|
|
166
|
+
.digest("hex");
|
|
167
|
+
return {
|
|
168
|
+
...item.record,
|
|
169
|
+
attemptId: "frontend-plan-coverage-reducer",
|
|
170
|
+
eventId: `coverage-reduced:${revisionId}`,
|
|
171
|
+
requestId: `coverage-reduced:${revisionId}`,
|
|
172
|
+
fact, payloadSha256,
|
|
173
|
+
};
|
|
174
|
+
}));
|
|
175
|
+
return merged;
|
|
176
|
+
}
|
|
177
|
+
export function normalizeParallelCoverageShardRecords(records, shardNumber) {
|
|
178
|
+
const prefix = `VT-SHARD-${shardNumber}-`;
|
|
179
|
+
// Models sometimes drop the `VT-` stem when applying the shard namespace
|
|
180
|
+
// (observed: frozen `VT-X` became `VT-SHARD-2-X`), so restoring the
|
|
181
|
+
// canonical id requires re-adding the stem after the strip.
|
|
182
|
+
const stripId = (id) => {
|
|
183
|
+
if (typeof id !== "string" || !id.startsWith(prefix))
|
|
184
|
+
return id;
|
|
185
|
+
const base = id.slice(prefix.length);
|
|
186
|
+
return base.startsWith("VT-") ? base : `VT-${base}`;
|
|
187
|
+
};
|
|
188
|
+
const stripIdList = (ids) => Array.isArray(ids) ? ids.map((id) => stripId(id)) : ids;
|
|
189
|
+
return records.map((record) => {
|
|
190
|
+
const fact = record.fact;
|
|
191
|
+
if (!fact || typeof fact !== "object")
|
|
192
|
+
return record;
|
|
193
|
+
const kind = fact.kind;
|
|
194
|
+
const entry = fact.entry;
|
|
195
|
+
if (!entry || typeof entry !== "object")
|
|
196
|
+
return record;
|
|
197
|
+
let rewritten;
|
|
198
|
+
const replaces = kind === "plan-verification-target" ? stripId(fact.replaces) : fact.replaces;
|
|
199
|
+
if (kind === "plan-verification-target") {
|
|
200
|
+
const id = stripId(entry.id);
|
|
201
|
+
if (id !== entry.id || replaces !== fact.replaces)
|
|
202
|
+
rewritten = { ...entry, id };
|
|
203
|
+
}
|
|
204
|
+
else if (kind === "plan-requirement") {
|
|
205
|
+
const verificationTargetIds = stripIdList(entry.verificationTargetIds);
|
|
206
|
+
if (verificationTargetIds !== entry.verificationTargetIds) {
|
|
207
|
+
rewritten = { ...entry, verificationTargetIds };
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
else if (kind === "state-flow") {
|
|
211
|
+
const rewriteBoundTargets = (item) => {
|
|
212
|
+
if (!item || typeof item !== "object")
|
|
213
|
+
return item;
|
|
214
|
+
return {
|
|
215
|
+
...item,
|
|
216
|
+
verificationTargetIds: stripIdList(item.verificationTargetIds),
|
|
217
|
+
};
|
|
218
|
+
};
|
|
219
|
+
const uiStates = Array.isArray(entry.uiStates)
|
|
220
|
+
? entry.uiStates.map(rewriteBoundTargets)
|
|
221
|
+
: entry.uiStates;
|
|
222
|
+
const interactions = Array.isArray(entry.interactions)
|
|
223
|
+
? entry.interactions.map(rewriteBoundTargets)
|
|
224
|
+
: entry.interactions;
|
|
225
|
+
if (uiStates !== entry.uiStates || interactions !== entry.interactions) {
|
|
226
|
+
rewritten = { ...entry, uiStates, interactions };
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
return rewritten
|
|
230
|
+
? {
|
|
231
|
+
...record,
|
|
232
|
+
fact: { ...fact, ...(replaces !== undefined ? { replaces } : {}), entry: rewritten },
|
|
233
|
+
// Keep the integrity hash consistent with the rewritten
|
|
234
|
+
// payload, or the adoption replay check reports the
|
|
235
|
+
// normalized record as a tampered source event.
|
|
236
|
+
payloadSha256: typedEventPayloadSha256({
|
|
237
|
+
...fact,
|
|
238
|
+
...(replaces !== undefined ? { replaces } : {}),
|
|
239
|
+
entry: rewritten,
|
|
240
|
+
}),
|
|
241
|
+
}
|
|
242
|
+
: record;
|
|
243
|
+
});
|
|
244
|
+
}
|
|
245
|
+
export function isPlannerThinkingExhausted(result, committedAnyFacts) {
|
|
246
|
+
if (result.ok)
|
|
247
|
+
return false;
|
|
248
|
+
const evidence = readWriterThinkingExhaustionEvidence(result);
|
|
249
|
+
if (evidence.stopReason !== "length")
|
|
250
|
+
return false;
|
|
251
|
+
if (evidence.thinkingObserved !== true)
|
|
252
|
+
return false;
|
|
253
|
+
if (committedAnyFacts)
|
|
254
|
+
return false;
|
|
255
|
+
if ((result.assistantText ?? "").trim())
|
|
256
|
+
return false;
|
|
257
|
+
// Gateways sometimes relabel a length-stopped stream as `network` or
|
|
258
|
+
// `nonzero-exit`; provider evidence outweighs the transport label.
|
|
259
|
+
if (result.failureCategory &&
|
|
260
|
+
!["empty-output", "network", "nonzero-exit", "unknown"].includes(result.failureCategory))
|
|
261
|
+
return false;
|
|
262
|
+
return true;
|
|
263
|
+
}
|
|
34
264
|
/**
|
|
35
265
|
* The writer session burned an excessive token budget (a read-edit-test loop
|
|
36
266
|
* that never converged) and still failed. Distinct from empty-output so the
|
|
@@ -69,8 +299,6 @@ function readWriterThinkingExhaustionEvidence(result) {
|
|
|
69
299
|
export function isWriterThinkingExhausted(result, mapped, changeManifestChangedFiles) {
|
|
70
300
|
if (mapped.ok)
|
|
71
301
|
return false;
|
|
72
|
-
if (mapped.failureCategory !== "empty-output")
|
|
73
|
-
return false;
|
|
74
302
|
const evidence = readWriterThinkingExhaustionEvidence(result);
|
|
75
303
|
if (evidence.stopReason !== "length")
|
|
76
304
|
return false;
|
|
@@ -78,6 +306,13 @@ export function isWriterThinkingExhausted(result, mapped, changeManifestChangedF
|
|
|
78
306
|
return false;
|
|
79
307
|
if ((evidence.writeToolCallCount ?? 0) !== 0)
|
|
80
308
|
return false;
|
|
309
|
+
// Gateways sometimes classify a length-stopped stream as `network` or
|
|
310
|
+
// `nonzero-exit` because the terminal event is carried in stderr. The
|
|
311
|
+
// provider evidence is stronger than that transport label when no write
|
|
312
|
+
// tool was called and the run produced no diff.
|
|
313
|
+
if (mapped.failureCategory &&
|
|
314
|
+
!["empty-output", "network", "nonzero-exit", "unknown"].includes(mapped.failureCategory))
|
|
315
|
+
return false;
|
|
81
316
|
if (changeManifestChangedFiles === undefined)
|
|
82
317
|
return false;
|
|
83
318
|
if (changeManifestChangedFiles.length !== 0)
|
|
@@ -217,6 +452,8 @@ export const FRONTEND_CONTRACT_RECORD_TOOL_NAMES = [
|
|
|
217
452
|
"record_handoff_intent",
|
|
218
453
|
"record_open_question",
|
|
219
454
|
"record_split_proposal",
|
|
455
|
+
"record_ui_state",
|
|
456
|
+
"record_required_deliverables",
|
|
220
457
|
];
|
|
221
458
|
export const FRONTEND_CONTRACT_TERMINAL_TOOL_NAMES = [
|
|
222
459
|
"finalize_contract",
|
|
@@ -230,6 +467,7 @@ export const FRONTEND_SCOUT_EVIDENCE_TOOL_NAMES = [
|
|
|
230
467
|
export const FRONTEND_PLAN_RECORD_TOOL_NAMES = [
|
|
231
468
|
"record_route_selection",
|
|
232
469
|
"record_component_choice",
|
|
470
|
+
"record_state_registry",
|
|
233
471
|
"record_state_flow",
|
|
234
472
|
"record_data_flow",
|
|
235
473
|
"record_mock_api",
|
|
@@ -873,8 +1111,111 @@ async function resolveFrontendPlanNewComponentSourceReferences(input) {
|
|
|
873
1111
|
}
|
|
874
1112
|
}
|
|
875
1113
|
/**
|
|
876
|
-
*
|
|
877
|
-
*
|
|
1114
|
+
* Frozen canonical requirements keyed by id, resolved from the source-fidelity
|
|
1115
|
+
* ledger before the contract node starts. record_requirement commits these
|
|
1116
|
+
* runtime-owned values so the model can never rewrite authoritative requirement
|
|
1117
|
+
* text or stringify the fragment bindings.
|
|
1118
|
+
*/
|
|
1119
|
+
async function resolveFrontendCanonicalRequirements(input) {
|
|
1120
|
+
const binding = input.sourceBinding;
|
|
1121
|
+
if (!binding || binding.schemaVersion !== 2 || !binding.ledgerPath) {
|
|
1122
|
+
return new Map();
|
|
1123
|
+
}
|
|
1124
|
+
const absolutePath = path.resolve(input.cwd, binding.ledgerPath);
|
|
1125
|
+
const workspaceRoot = path.resolve(input.cwd);
|
|
1126
|
+
if (absolutePath !== workspaceRoot &&
|
|
1127
|
+
!absolutePath.startsWith(`${workspaceRoot}${path.sep}`)) {
|
|
1128
|
+
throw new Error("canonical requirement ledger escapes workspace");
|
|
1129
|
+
}
|
|
1130
|
+
try {
|
|
1131
|
+
const raw = await readFile(absolutePath, "utf8");
|
|
1132
|
+
if (sha256Text(raw) !== binding.ledgerSha256) {
|
|
1133
|
+
throw new Error("source ledger is stale");
|
|
1134
|
+
}
|
|
1135
|
+
const ledger = parseLedgerJson(raw);
|
|
1136
|
+
const sourceFragments = new Map();
|
|
1137
|
+
const boundFragments = new Map(ledger.fragments.map((fragment) => [fragment.id, fragment]));
|
|
1138
|
+
for (const sourcePath of new Set(ledger.fragments.map((fragment) => fragment.path))) {
|
|
1139
|
+
const sourceAbsolute = path.resolve(workspaceRoot, sourcePath);
|
|
1140
|
+
if (!sourceAbsolute.startsWith(`${workspaceRoot}${path.sep}`)) {
|
|
1141
|
+
throw new Error("source fragment escapes workspace");
|
|
1142
|
+
}
|
|
1143
|
+
const content = await readFile(sourceAbsolute, "utf8");
|
|
1144
|
+
for (const fragment of splitDocumentIntoFragments({ path: sourcePath, content })) {
|
|
1145
|
+
const frozen = boundFragments.get(fragment.id);
|
|
1146
|
+
if (frozen && frozen.sha256 === sha256Text(fragment.text)) {
|
|
1147
|
+
sourceFragments.set(fragment.id, fragment.text);
|
|
1148
|
+
}
|
|
1149
|
+
}
|
|
1150
|
+
}
|
|
1151
|
+
return new Map(ledger.canonicalRequirements.map((requirement) => [
|
|
1152
|
+
requirement.id,
|
|
1153
|
+
{
|
|
1154
|
+
text: requirement.text,
|
|
1155
|
+
sourceFragmentIds: [...(requirement.sourceFragmentIds ?? [])],
|
|
1156
|
+
sourceFragments,
|
|
1157
|
+
},
|
|
1158
|
+
]));
|
|
1159
|
+
}
|
|
1160
|
+
catch (error) {
|
|
1161
|
+
throw new Error(`canonical requirement sources unavailable: ${error instanceof Error ? error.message : String(error)}`);
|
|
1162
|
+
}
|
|
1163
|
+
}
|
|
1164
|
+
/**
|
|
1165
|
+
* Authoritative UI state ids declared by the contract node
|
|
1166
|
+
* (ui-state-declaration facts). Empty when the source declares no UI-state
|
|
1167
|
+
* table, in which case the plan's state vocabulary is registry-only.
|
|
1168
|
+
*/
|
|
1169
|
+
async function resolveFrontendDeclaredUiStateIds(input) {
|
|
1170
|
+
try {
|
|
1171
|
+
const records = await readTypedEventStoreFromJsonl(path.join(input.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl"));
|
|
1172
|
+
return records.flatMap((record) => {
|
|
1173
|
+
const fact = record.fact;
|
|
1174
|
+
return record.phase === "committed" &&
|
|
1175
|
+
fact.kind === "ui-state-declaration" &&
|
|
1176
|
+
typeof fact.id === "string"
|
|
1177
|
+
? [fact.id]
|
|
1178
|
+
: [];
|
|
1179
|
+
});
|
|
1180
|
+
}
|
|
1181
|
+
catch {
|
|
1182
|
+
// No contract facts (or no declared states): registry-only vocabulary.
|
|
1183
|
+
return [];
|
|
1184
|
+
}
|
|
1185
|
+
}
|
|
1186
|
+
/**
|
|
1187
|
+
* Resolve the frozen canonical behavior verification-target ids the PRD
|
|
1188
|
+
* declares. The requirement text is the same authority the design review reads
|
|
1189
|
+
* when it rejects `VT-SHARD-N-…` / variant ids as a contract-requirement gap,
|
|
1190
|
+
* so extracting `VT-…` tokens from the frozen contract requirement facts makes
|
|
1191
|
+
* that authority deterministic instead of prose-only. An empty result (no PRD
|
|
1192
|
+
* declared any behavior target id) disables the canonical check and keeps the
|
|
1193
|
+
* historical free-form path.
|
|
1194
|
+
*/
|
|
1195
|
+
export async function resolveFrontendCanonicalVerificationTargetIds(input) {
|
|
1196
|
+
try {
|
|
1197
|
+
const records = await readTypedEventStoreFromJsonl(path.join(input.runDir, "frontend-contract-pi", "contract-typed-facts.jsonl"));
|
|
1198
|
+
const ids = new Set();
|
|
1199
|
+
for (const record of records) {
|
|
1200
|
+
const fact = record.fact;
|
|
1201
|
+
if (record.phase !== "committed" || fact.kind !== "requirement") {
|
|
1202
|
+
continue;
|
|
1203
|
+
}
|
|
1204
|
+
const text = typeof fact.text === "string" ? fact.text : "";
|
|
1205
|
+
for (const match of text.matchAll(/\bVT-[A-Z0-9][A-Z0-9_-]*\b/g)) {
|
|
1206
|
+
ids.add(match[0]);
|
|
1207
|
+
}
|
|
1208
|
+
}
|
|
1209
|
+
return [...ids].sort();
|
|
1210
|
+
}
|
|
1211
|
+
catch {
|
|
1212
|
+
// No contract facts (or unreadable): fall back to no canonical set.
|
|
1213
|
+
return [];
|
|
1214
|
+
}
|
|
1215
|
+
}
|
|
1216
|
+
/**
|
|
1217
|
+
* A+B: `frontend-plan-pi` records its decision ledger through incremental
|
|
1218
|
+
* `record_*` tools (origin=plan) and closes with exactly one
|
|
878
1219
|
* `finalize_plan` terminal. A later attempt may explicitly adopt a quarantined
|
|
879
1220
|
* fact via `adopt_staged_fact`. Flush writes `plan-typed-facts.jsonl` for the
|
|
880
1221
|
* node validator / compile authority.
|
|
@@ -965,23 +1306,24 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
965
1306
|
paths: stringArray,
|
|
966
1307
|
conflicts: stringArray,
|
|
967
1308
|
}, { additionalProperties: false });
|
|
968
|
-
//
|
|
969
|
-
//
|
|
970
|
-
//
|
|
971
|
-
// —
|
|
972
|
-
|
|
973
|
-
|
|
1309
|
+
// Verification mode is runtime-owned (contract v2): the plan submits a
|
|
1310
|
+
// commandId referencing the frozen command directory, never a type. A
|
|
1311
|
+
// soft Type.String would let the model commit values that only fail at
|
|
1312
|
+
// compile time — keep the reference a required string and validate it
|
|
1313
|
+
// against the directory at the tool boundary below.
|
|
1314
|
+
const verificationTargetScopeSchema = Type.Union([
|
|
974
1315
|
Type.Literal("unit"),
|
|
975
1316
|
Type.Literal("component"),
|
|
976
1317
|
Type.Literal("integration"),
|
|
977
|
-
Type.Literal("mock"),
|
|
978
1318
|
]);
|
|
979
1319
|
const verificationTargetSchema = Type.Object({
|
|
980
1320
|
id: Type.String({}),
|
|
981
|
-
|
|
982
|
-
|
|
1321
|
+
commandId: Type.String({
|
|
1322
|
+
description: "Frozen command directory key (e.g. verify-npm-run-build); the runtime resolves mode and label from it.",
|
|
1323
|
+
}),
|
|
983
1324
|
file: Type.String({}),
|
|
984
1325
|
requirementIds: stringArray,
|
|
1326
|
+
scope: Type.Optional(verificationTargetScopeSchema),
|
|
985
1327
|
uiStates: Type.Optional(Type.Array(Type.String({}), {
|
|
986
1328
|
description: "Optional only at this tool boundary. An omitted value is deterministically recorded as []. Pass an explicit array for new calls.",
|
|
987
1329
|
})),
|
|
@@ -1007,6 +1349,12 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1007
1349
|
line: Type.Optional(Type.Number({})),
|
|
1008
1350
|
}, { additionalProperties: false })),
|
|
1009
1351
|
rationale: Type.String({}),
|
|
1352
|
+
covers: Type.Optional(Type.Array(Type.String({}), {
|
|
1353
|
+
description: "UI state and/or interaction names this single component choice covers (one choice may cover many ids).",
|
|
1354
|
+
})),
|
|
1355
|
+
evidencePath: Type.Optional(Type.String({
|
|
1356
|
+
description: "REQUIRED for decision=reuse-existing: repo-relative path whose existing file is the reuse evidence. Greenfield paths must use decision=new.",
|
|
1357
|
+
})),
|
|
1010
1358
|
}, { additionalProperties: false });
|
|
1011
1359
|
const stringList = (value) => Array.isArray(value)
|
|
1012
1360
|
? value.filter((item) => typeof item === "string")
|
|
@@ -1085,7 +1433,7 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1085
1433
|
const recordComponentChoiceTool = defineTool({
|
|
1086
1434
|
name: "record_component_choice",
|
|
1087
1435
|
label: "record_component_choice",
|
|
1088
|
-
description: "Record ONE component choice (origin=plan component-choice fact).
|
|
1436
|
+
description: "Record ONE component choice (origin=plan component-choice fact). One choice may cover MULTIPLE UI states/interactions via covers: [\"<state-or-interaction name>\", ...]; do not emit one row per interaction. For decision=new, pass sourceRequirementIds containing the frozen requirement ID(s) that mandate the component and sourceFragmentId selecting one frozen citation listed in the plan checklist; the runtime validates the relation and derives the exact PRD specReference. Do not read the PRD or invent a path/line. decision=reuse-existing is only for components that already exist in the repo and REQUIRES evidencePath: a repo-relative path to the existing file that proves the reuse — the runtime verifies the file exists (fresh evidence); a path with no existing file is greenfield and must use decision=new instead. Call up to 5 component choices per assistant message; never more than 5 per message. Optionally include stylingStrategy (set it once, on the first call). Example: {\"choice\": {\"purpose\": \"<interaction or UI state name>\", \"component\": \"<component name>\", \"decision\": \"new\", \"covers\": [\"<other interaction names this component also serves>\"]}, \"sourceRequirementIds\": [\"<AC-XXX mandating this component>\"], \"sourceFragmentId\": \"<REQ-SRC-...>\"} or {\"choice\": {\"purpose\": \"FocusQueuePanel\", \"component\": \"FocusQueuePanel\", \"decision\": \"reuse-existing\", \"evidencePath\": \"src/journal.ts\"}}",
|
|
1089
1437
|
promptSnippet: "Record 1-5 component choices (up to 5 per message).",
|
|
1090
1438
|
parameters: Type.Object({
|
|
1091
1439
|
choice: uiComponentChoiceSchema,
|
|
@@ -1136,6 +1484,46 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1136
1484
|
...(citation.line !== undefined ? { line: citation.line } : {}),
|
|
1137
1485
|
};
|
|
1138
1486
|
}
|
|
1487
|
+
if (choice.decision === "reuse-existing") {
|
|
1488
|
+
// reuse-existing must cite a file that actually exists NOW (fresh
|
|
1489
|
+
// existence evidence, same semantics as Scout pathEvidence.fresh).
|
|
1490
|
+
// A path with no file on disk is greenfield: decision=new is the
|
|
1491
|
+
// only honest choice (dogfood run dag-1788504923861-0f7b17a9
|
|
1492
|
+
// claimed 18 reuse-existing conventions in a not-yet-written
|
|
1493
|
+
// src/planner.ts and every gate let it through).
|
|
1494
|
+
const evidencePath = typeof choice.evidencePath === "string"
|
|
1495
|
+
? choice.evidencePath.trim()
|
|
1496
|
+
: "";
|
|
1497
|
+
if (!evidencePath) {
|
|
1498
|
+
return planToolReceipt({
|
|
1499
|
+
ok: false,
|
|
1500
|
+
kind: "component-choice",
|
|
1501
|
+
error: "decision=reuse-existing requires evidencePath naming the existing repo file that proves the reuse; if the file does not exist yet, use decision=new",
|
|
1502
|
+
});
|
|
1503
|
+
}
|
|
1504
|
+
if (input.workspaceRoot) {
|
|
1505
|
+
const absolute = path.resolve(input.workspaceRoot, evidencePath);
|
|
1506
|
+
const workspaceRoot = path.resolve(input.workspaceRoot);
|
|
1507
|
+
const contained = absolute === workspaceRoot ||
|
|
1508
|
+
absolute.startsWith(`${workspaceRoot}${path.sep}`);
|
|
1509
|
+
let exists = false;
|
|
1510
|
+
if (contained) {
|
|
1511
|
+
try {
|
|
1512
|
+
exists = (await stat(absolute)).isFile();
|
|
1513
|
+
}
|
|
1514
|
+
catch {
|
|
1515
|
+
exists = false;
|
|
1516
|
+
}
|
|
1517
|
+
}
|
|
1518
|
+
if (!exists) {
|
|
1519
|
+
return planToolReceipt({
|
|
1520
|
+
ok: false,
|
|
1521
|
+
kind: "component-choice",
|
|
1522
|
+
error: `decision=reuse-existing evidencePath "${evidencePath}" has no fresh existence evidence (file not found in the workspace); reuse requires an existing file — use decision=new for greenfield paths`,
|
|
1523
|
+
});
|
|
1524
|
+
}
|
|
1525
|
+
}
|
|
1526
|
+
}
|
|
1139
1527
|
const components = typeof choice.component === "string" ? [choice.component] : [];
|
|
1140
1528
|
const result = await adoptPlanFact("component-choice", `${attemptId}:record_component_choice:${randomUUID()}`, {
|
|
1141
1529
|
kind: "component-choice",
|
|
@@ -1159,10 +1547,86 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1159
1547
|
return planToolReceipt(echo);
|
|
1160
1548
|
},
|
|
1161
1549
|
});
|
|
1550
|
+
// Global UX vocabulary: one compact registry committed BEFORE any state
|
|
1551
|
+
// flow details. Coverage may slice by AC; UX must not — the registry is
|
|
1552
|
+
// the anti-duplication anchor that keeps every later slice on the same
|
|
1553
|
+
// named concepts instead of re-inventing them per AC chunk.
|
|
1554
|
+
const recordStateRegistryTool = defineTool({
|
|
1555
|
+
name: "record_state_registry",
|
|
1556
|
+
label: "record_state_registry",
|
|
1557
|
+
description: "Commit the GLOBAL UX vocabulary (origin=plan state-registry fact) BEFORE any record_state_flow call: uiStateNames (use the contract's declared authoritative state ids when provided) and interactionNames (stable behavior-domain kebab-case names, e.g. planner-task-edit / focus-queue-move — one name per behavior domain, never one per AC). Empty arrays explicitly mean the request has no UI state or interaction vocabulary. Re-record the full vocabulary to correct it; the latest commit wins, but it may not remove names still referenced by committed state-flow facts. Example: {\"uiStateNames\": [\"planner-empty\"], \"interactionNames\": [\"planner-task-create\", \"focus-queue-move\"]}",
|
|
1558
|
+
promptSnippet: "Record the global UI-state/interaction vocabulary once, before any state flow.",
|
|
1559
|
+
parameters: Type.Object({
|
|
1560
|
+
uiStateNames: stringArray,
|
|
1561
|
+
interactionNames: stringArray,
|
|
1562
|
+
}, { additionalProperties: false }),
|
|
1563
|
+
async execute(_toolCallId, params) {
|
|
1564
|
+
const uiStateNames = [...new Set(stringList(params?.uiStateNames).map((name) => name.trim()))];
|
|
1565
|
+
const interactionNames = [
|
|
1566
|
+
...new Set(stringList(params?.interactionNames).map((name) => name.trim())),
|
|
1567
|
+
];
|
|
1568
|
+
if (uiStateNames.some((name) => name.length === 0) || interactionNames.some((name) => name.length === 0)) {
|
|
1569
|
+
return planToolReceipt({
|
|
1570
|
+
ok: false,
|
|
1571
|
+
kind: "state-registry",
|
|
1572
|
+
error: "record_state_registry names must be non-empty strings",
|
|
1573
|
+
});
|
|
1574
|
+
}
|
|
1575
|
+
const invalidInteractionNames = interactionNames.filter((name) => !/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(name));
|
|
1576
|
+
if (invalidInteractionNames.length > 0) {
|
|
1577
|
+
return planToolReceipt({
|
|
1578
|
+
ok: false,
|
|
1579
|
+
kind: "state-registry",
|
|
1580
|
+
error: `record_state_registry interactionNames must be stable kebab-case behavior-domain names: ${invalidInteractionNames.join(", ")}`,
|
|
1581
|
+
});
|
|
1582
|
+
}
|
|
1583
|
+
if (input.declaredUiStateIds &&
|
|
1584
|
+
input.declaredUiStateIds.length > 0) {
|
|
1585
|
+
const declared = new Set(input.declaredUiStateIds);
|
|
1586
|
+
const undeclared = uiStateNames.filter((name) => !declared.has(name));
|
|
1587
|
+
if (undeclared.length > 0) {
|
|
1588
|
+
return planToolReceipt({
|
|
1589
|
+
ok: false,
|
|
1590
|
+
kind: "state-registry",
|
|
1591
|
+
error: `record_state_registry uiStateNames are not declared by the contract's authoritative UI-state table: ${undeclared.join(", ")}; use the declared ids (${input.declaredUiStateIds.join(", ")})`,
|
|
1592
|
+
});
|
|
1593
|
+
}
|
|
1594
|
+
const missing = input.declaredUiStateIds.filter((name) => !uiStateNames.includes(name));
|
|
1595
|
+
if (missing.length > 0) {
|
|
1596
|
+
return planToolReceipt({
|
|
1597
|
+
ok: false,
|
|
1598
|
+
kind: "state-registry",
|
|
1599
|
+
error: `record_state_registry must include every state from the contract's authoritative UI-state table; missing: ${missing.join(", ")}`,
|
|
1600
|
+
});
|
|
1601
|
+
}
|
|
1602
|
+
}
|
|
1603
|
+
// A correction is deterministic only when the new last-wins registry
|
|
1604
|
+
// still contains every live name already committed by state-flow facts.
|
|
1605
|
+
// This permits adding a missed concept, while preventing a registry edit
|
|
1606
|
+
// from retroactively orphaning earlier slices.
|
|
1607
|
+
const liveNames = collectCanonicalStateFlowNames(readCommittedEvents(store, attemptId));
|
|
1608
|
+
const orphanedUiStates = [...liveNames.uiStateNames].filter((name) => !uiStateNames.includes(name));
|
|
1609
|
+
const orphanedInteractions = [...liveNames.interactionNames].filter((name) => !interactionNames.includes(name));
|
|
1610
|
+
if (orphanedUiStates.length > 0 || orphanedInteractions.length > 0) {
|
|
1611
|
+
return planToolReceipt({
|
|
1612
|
+
ok: false,
|
|
1613
|
+
kind: "state-registry",
|
|
1614
|
+
error: `record_state_registry cannot remove names still referenced by committed state-flow facts (uiStates: ${orphanedUiStates.join(", ") || "none"}; interactions: ${orphanedInteractions.join(", ") || "none"}); first correct/remove those state-flow entries, then re-record the full registry`,
|
|
1615
|
+
});
|
|
1616
|
+
}
|
|
1617
|
+
const result = await adoptPlanFact("state-registry", `${attemptId}:record_state_registry:${randomUUID()}`, {
|
|
1618
|
+
kind: "state-registry",
|
|
1619
|
+
origin: "plan",
|
|
1620
|
+
uiStateNames,
|
|
1621
|
+
interactionNames,
|
|
1622
|
+
});
|
|
1623
|
+
return planToolReceipt(result);
|
|
1624
|
+
},
|
|
1625
|
+
});
|
|
1162
1626
|
const recordStateFlowTool = defineTool({
|
|
1163
1627
|
name: "record_state_flow",
|
|
1164
1628
|
label: "record_state_flow",
|
|
1165
|
-
description: "Record UI states and interactions as an origin=plan state-flow fact. To correct stale named entries from an earlier UX batch, include removeUiStateNames and/or removeInteractionNames. Example: {\"uiStates\": [{\"name\": \"<state>\", \"applicable\": true, \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}], \"interactions\": [{\"name\": \"<interaction>\", \"trigger\": \"<user event>\", \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}]}",
|
|
1629
|
+
description: "Record UI states and interactions as an origin=plan state-flow fact. REQUIRES a committed record_state_registry vocabulary first, and every name here must be in that registry; uiState names must also be contract-declared authoritative ids when the contract declares them. To correct stale named entries from an earlier UX batch, include removeUiStateNames and/or removeInteractionNames. Example: {\"uiStates\": [{\"name\": \"<state>\", \"applicable\": true, \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}], \"interactions\": [{\"name\": \"<interaction>\", \"trigger\": \"<user event>\", \"expectedBehavior\": \"<behavior>\", \"implementationTargets\": [\"<file>\"], \"verificationTargetIds\": [\"<VT-XXX>\"]}]}",
|
|
1166
1630
|
promptSnippet: "Record the plan state-flow fact.",
|
|
1167
1631
|
parameters: Type.Object({
|
|
1168
1632
|
uiStates: Type.Array(uiStateSchema),
|
|
@@ -1266,9 +1730,63 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1266
1730
|
}
|
|
1267
1731
|
interactions.push({ ...interaction, name: resolvedName });
|
|
1268
1732
|
}
|
|
1733
|
+
// UX vocabulary gate: every recorded state/interaction name must be
|
|
1734
|
+
// declared in the committed global registry first. This keeps UX out
|
|
1735
|
+
// of AC-number slicing — the model commits one compact vocabulary
|
|
1736
|
+
// (record_state_registry), then fills behavior-domain details, and a
|
|
1737
|
+
// later slice cannot silently rename an earlier concept (dogfood
|
|
1738
|
+
// dag-1788504923861-0f7b17a9: 5 renames of "prioritize" + 7 of
|
|
1739
|
+
// "focus queue" across AC chunks).
|
|
1269
1740
|
const states = uiStates
|
|
1270
1741
|
.map((state) => (typeof state?.name === "string" ? state.name : ""))
|
|
1271
1742
|
.filter(Boolean);
|
|
1743
|
+
const committedRegistry = readCommittedEvents(store, attemptId)
|
|
1744
|
+
.map((event) => event.fact)
|
|
1745
|
+
.filter((fact) => isRecordObject(fact) && fact.kind === "state-registry")
|
|
1746
|
+
.at(-1);
|
|
1747
|
+
if (!committedRegistry) {
|
|
1748
|
+
return planToolReceipt({
|
|
1749
|
+
ok: false,
|
|
1750
|
+
kind: "state-flow",
|
|
1751
|
+
error: "record_state_flow requires a committed UX registry first: call record_state_registry with the full uiStateNames/interactionNames vocabulary, then record state flows against it",
|
|
1752
|
+
});
|
|
1753
|
+
}
|
|
1754
|
+
const registryUiStateNames = new Set(stringList(committedRegistry.uiStateNames));
|
|
1755
|
+
const registryInteractionNames = new Set(stringList(committedRegistry.interactionNames));
|
|
1756
|
+
const unknownStates = states.filter((name) => !registryUiStateNames.has(name));
|
|
1757
|
+
if (unknownStates.length > 0) {
|
|
1758
|
+
return planToolReceipt({
|
|
1759
|
+
ok: false,
|
|
1760
|
+
kind: "state-flow",
|
|
1761
|
+
error: `record_state_flow uiState names not in the committed registry: ${unknownStates.join(", ")}; re-record record_state_registry with the complete vocabulary first (registered: ${[...registryUiStateNames].join(", ") || "(none)"})`,
|
|
1762
|
+
});
|
|
1763
|
+
}
|
|
1764
|
+
const unknownInteractions = interactions
|
|
1765
|
+
.map((interaction) => typeof interaction?.name === "string" ? interaction.name : "")
|
|
1766
|
+
.filter((name) => name && !registryInteractionNames.has(name));
|
|
1767
|
+
if (unknownInteractions.length > 0) {
|
|
1768
|
+
return planToolReceipt({
|
|
1769
|
+
ok: false,
|
|
1770
|
+
kind: "state-flow",
|
|
1771
|
+
error: `record_state_flow interaction names not in the committed registry: ${unknownInteractions.join(", ")}; re-record record_state_registry with the complete vocabulary first (registered: ${[...registryInteractionNames].join(", ") || "(none)"})`,
|
|
1772
|
+
});
|
|
1773
|
+
}
|
|
1774
|
+
// Authoritative UI states: when the contract node declared the
|
|
1775
|
+
// source's UI-state table, planner states must bind those ids — no
|
|
1776
|
+
// invented variants (planner-empty/create-invalid/no-results/
|
|
1777
|
+
// editing/focus-full/storage-unavailable, not "planner-list").
|
|
1778
|
+
if (input.declaredUiStateIds &&
|
|
1779
|
+
input.declaredUiStateIds.length > 0) {
|
|
1780
|
+
const declared = new Set(input.declaredUiStateIds);
|
|
1781
|
+
const undeclaredStates = states.filter((name) => !declared.has(name));
|
|
1782
|
+
if (undeclaredStates.length > 0) {
|
|
1783
|
+
return planToolReceipt({
|
|
1784
|
+
ok: false,
|
|
1785
|
+
kind: "state-flow",
|
|
1786
|
+
error: `record_state_flow uiState names are not declared by the contract's authoritative UI-state table: ${undeclaredStates.join(", ")}; use the declared ids (${input.declaredUiStateIds.join(", ")}), or record a design deviation if the source table is genuinely incomplete`,
|
|
1787
|
+
});
|
|
1788
|
+
}
|
|
1789
|
+
}
|
|
1272
1790
|
const result = await adoptPlanFact("state-flow", `${attemptId}:record_state_flow:${randomUUID()}`, {
|
|
1273
1791
|
kind: "state-flow",
|
|
1274
1792
|
origin: "plan",
|
|
@@ -1427,6 +1945,15 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1427
1945
|
});
|
|
1428
1946
|
}
|
|
1429
1947
|
}
|
|
1948
|
+
if (id &&
|
|
1949
|
+
activeRequirementScope.length > 0 &&
|
|
1950
|
+
!activeRequirementScope.includes(id)) {
|
|
1951
|
+
return planToolReceipt({
|
|
1952
|
+
ok: false,
|
|
1953
|
+
kind: "plan-requirement",
|
|
1954
|
+
error: `record_plan_requirement id "${id}" is outside this session's requirement scope [${activeRequirementScope.join(", ")}]`,
|
|
1955
|
+
});
|
|
1956
|
+
}
|
|
1430
1957
|
const result = await adoptPlanFact("plan-requirement", `${attemptId}:record_plan_requirement:${randomUUID()}`, {
|
|
1431
1958
|
kind: "plan-requirement",
|
|
1432
1959
|
origin: "plan",
|
|
@@ -1439,7 +1966,11 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1439
1966
|
const recordPlanVerificationTargetTool = defineTool({
|
|
1440
1967
|
name: "record_plan_verification_target",
|
|
1441
1968
|
label: "record_plan_verification_target",
|
|
1442
|
-
description: "Commit one plan verification target entry (origin=plan plan-verification-target fact).
|
|
1969
|
+
description: "Commit one plan verification target entry (origin=plan plan-verification-target fact). Reference a frozen verification command by commandId (see the frozen command directory in your prompt: static commands are project-wide checks traced by file and command only; behavior commands need a test file whose describe/it/test title contains the target id). For behavior targets, call once per distinct behavior, not mechanically once per requirement: one target may cover multiple related requirementIds. For a correction, re-submit the same id with replace=true; the ledger compiles the latest replacement. A behavior target id is the stable trace token that implementation must place in a real describe/it/test title. Entry carries id, commandId, file, requirementIds, and uiStates; optional scope (unit | component | integration) is display-only. Free-form symbol text is not accepted. IMPORTANT: batch up to 4 record_* calls per assistant message; never batch more than 4 — a larger single message risks output truncation under a small output window. Example: {\"entry\": {\"id\": \"VT-DASHBOARD-SHELL\", \"commandId\": \"<frozen behavior command id>\", \"file\": \"<test file>\", \"requirementIds\": [\"AC-001\", \"AC-002\"], \"uiStates\": []}}" +
|
|
1970
|
+
(input.canonicalVerificationTargetIds &&
|
|
1971
|
+
input.canonicalVerificationTargetIds.length > 0
|
|
1972
|
+
? ` Frozen canonical behavior target ids (use exactly for behavior targets): ${input.canonicalVerificationTargetIds.join(", ")}.`
|
|
1973
|
+
: ""),
|
|
1443
1974
|
promptSnippet: "Commit 1-4 plan verification target entries (up to 4 per message).",
|
|
1444
1975
|
parameters: Type.Object({
|
|
1445
1976
|
entry: verificationTargetSchema,
|
|
@@ -1454,19 +1985,61 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1454
1985
|
error: "record_plan_verification_target requires a non-empty entry object",
|
|
1455
1986
|
});
|
|
1456
1987
|
}
|
|
1457
|
-
//
|
|
1458
|
-
//
|
|
1459
|
-
//
|
|
1460
|
-
//
|
|
1461
|
-
|
|
1462
|
-
|
|
1463
|
-
|
|
1988
|
+
// Contract v2: the plan references a frozen command by commandId;
|
|
1989
|
+
// mode and label are resolved by the runtime from the frozen
|
|
1990
|
+
// command directory (run.json frontend-verify-shell bundle lanes).
|
|
1991
|
+
// Reject unknown ids and behavior commands bound to non-test
|
|
1992
|
+
// files here so the model fixes them in-node instead of the
|
|
1993
|
+
// attempt dying at materialization or at the writer focused-check.
|
|
1994
|
+
const verificationCommandId = typeof rawEntry.commandId === "string"
|
|
1995
|
+
? rawEntry.commandId.trim()
|
|
1996
|
+
: "";
|
|
1997
|
+
if (!verificationCommandId) {
|
|
1464
1998
|
return planToolReceipt({
|
|
1465
1999
|
ok: false,
|
|
1466
2000
|
kind: "plan-verification-target",
|
|
1467
|
-
error: `record_plan_verification_target entry.
|
|
2001
|
+
error: `record_plan_verification_target entry.commandId is required (received ${JSON.stringify(rawEntry.commandId ?? null)}); pick one id from the frozen command directory in your prompt`,
|
|
1468
2002
|
});
|
|
1469
2003
|
}
|
|
2004
|
+
const { deriveFrontendVerifyCommandDirectoryFromRun, isFrontendTestFilePath, } = await import("../workflows/dag/frontend-implementation-contract.js");
|
|
2005
|
+
const verifyDirectory = await deriveFrontendVerifyCommandDirectoryFromRun(input.runDir);
|
|
2006
|
+
if (verifyDirectory.length > 0) {
|
|
2007
|
+
const directoryEntry = verifyDirectory.find((entry) => entry.commandId === verificationCommandId);
|
|
2008
|
+
if (!directoryEntry) {
|
|
2009
|
+
return planToolReceipt({
|
|
2010
|
+
ok: false,
|
|
2011
|
+
kind: "plan-verification-target",
|
|
2012
|
+
error: `record_plan_verification_target verification-target-unknown-command: unknown commandId "${verificationCommandId}"; available frozen commands: [${verifyDirectory.map((entry) => `${entry.commandId} (${entry.mode}: ${entry.label})`).join(", ")}]`,
|
|
2013
|
+
});
|
|
2014
|
+
}
|
|
2015
|
+
if (directoryEntry.mode === "behavior" &&
|
|
2016
|
+
typeof rawEntry.file === "string" &&
|
|
2017
|
+
!isFrontendTestFilePath(rawEntry.file)) {
|
|
2018
|
+
return planToolReceipt({
|
|
2019
|
+
ok: false,
|
|
2020
|
+
kind: "plan-verification-target",
|
|
2021
|
+
error: `record_plan_verification_target verification-target-phase-mismatch: behavior command "${directoryEntry.label}" (${directoryEntry.commandId}) must bind a test file (__tests__/, tests?/, e2e/, cypress/, *.test.*, *.spec.*, *.cy.*); received file "${rawEntry.file}"`,
|
|
2022
|
+
});
|
|
2023
|
+
}
|
|
2024
|
+
// Canonical-identity check for behavior targets: the PRD freezes
|
|
2025
|
+
// the exact behavior verification-target ids (e.g.
|
|
2026
|
+
// VT-SMOKE-COUNTER-BEHAVIOR). A committed non-canonical id is
|
|
2027
|
+
// immutable and design review rejects it as a
|
|
2028
|
+
// contract-requirement gap, so reject invented ids here.
|
|
2029
|
+
if (directoryEntry.mode === "behavior" &&
|
|
2030
|
+
input.canonicalVerificationTargetIds &&
|
|
2031
|
+
input.canonicalVerificationTargetIds.length > 0) {
|
|
2032
|
+
const canonicalTargetId = typeof rawEntry.id === "string" ? rawEntry.id.trim() : "";
|
|
2033
|
+
if (canonicalTargetId &&
|
|
2034
|
+
!input.canonicalVerificationTargetIds.includes(canonicalTargetId)) {
|
|
2035
|
+
return planToolReceipt({
|
|
2036
|
+
ok: false,
|
|
2037
|
+
kind: "plan-verification-target",
|
|
2038
|
+
error: `record_plan_verification_target id "${canonicalTargetId}" is not a frozen canonical behavior verification target; canonical ids are: ${input.canonicalVerificationTargetIds.join(", ")}`,
|
|
2039
|
+
});
|
|
2040
|
+
}
|
|
2041
|
+
}
|
|
2042
|
+
}
|
|
1470
2043
|
// Duplicate-id rejection: committed typed facts are immutable, so
|
|
1471
2044
|
// re-recording the same VT id would deadlock the compile by default.
|
|
1472
2045
|
// replace=true is the explicit in-node correction path.
|
|
@@ -1535,6 +2108,15 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1535
2108
|
: [];
|
|
1536
2109
|
}));
|
|
1537
2110
|
const unknownRequirementIds = stringList(rawEntry.requirementIds).filter((id) => !declaredRequirementIds.has(id));
|
|
2111
|
+
const outOfScopeRequirementIds = stringList(rawEntry.requirementIds).filter((id) => activeRequirementScope.length > 0 &&
|
|
2112
|
+
!activeRequirementScope.includes(id));
|
|
2113
|
+
if (outOfScopeRequirementIds.length > 0) {
|
|
2114
|
+
return planToolReceipt({
|
|
2115
|
+
ok: false,
|
|
2116
|
+
kind: "plan-verification-target",
|
|
2117
|
+
error: `record_plan_verification_target references requirements outside this session's scope [${activeRequirementScope.join(", ")}]: ${outOfScopeRequirementIds.join(", ")}`,
|
|
2118
|
+
});
|
|
2119
|
+
}
|
|
1538
2120
|
if (unknownRequirementIds.length > 0) {
|
|
1539
2121
|
return planToolReceipt({
|
|
1540
2122
|
ok: false,
|
|
@@ -1581,6 +2163,18 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1581
2163
|
error: "record_plan_evidence_gap requires a non-empty description describing the gap",
|
|
1582
2164
|
});
|
|
1583
2165
|
}
|
|
2166
|
+
const requirementId = typeof entry.requirementId === "string"
|
|
2167
|
+
? entry.requirementId
|
|
2168
|
+
: undefined;
|
|
2169
|
+
if (requirementId &&
|
|
2170
|
+
activeRequirementScope.length > 0 &&
|
|
2171
|
+
!activeRequirementScope.includes(requirementId)) {
|
|
2172
|
+
return planToolReceipt({
|
|
2173
|
+
ok: false,
|
|
2174
|
+
kind: "plan-evidence-gap",
|
|
2175
|
+
error: `record_plan_evidence_gap requirementId "${requirementId}" is outside this session's requirement scope [${activeRequirementScope.join(", ")}]`,
|
|
2176
|
+
});
|
|
2177
|
+
}
|
|
1584
2178
|
const result = await adoptPlanFact("plan-evidence-gap", `${attemptId}:record_plan_evidence_gap:${randomUUID()}`, { kind: "plan-evidence-gap", origin: "plan", entry });
|
|
1585
2179
|
return planToolReceipt(result);
|
|
1586
2180
|
},
|
|
@@ -1613,6 +2207,29 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1613
2207
|
? { realIntegrationGap: params.realIntegrationGap }
|
|
1614
2208
|
: {}),
|
|
1615
2209
|
};
|
|
2210
|
+
const latestRegistry = [...committed]
|
|
2211
|
+
.reverse()
|
|
2212
|
+
.map((record) => record.fact)
|
|
2213
|
+
.find((fact) => isRecordObject(fact) && fact.kind === "state-registry");
|
|
2214
|
+
if (latestRegistry) {
|
|
2215
|
+
const registryUiStates = new Set(stringList(latestRegistry.uiStateNames));
|
|
2216
|
+
const registryInteractions = new Set(stringList(latestRegistry.interactionNames));
|
|
2217
|
+
const live = collectCanonicalStateFlowNames(committed);
|
|
2218
|
+
const missingUiStates = [...registryUiStates].filter((name) => !live.uiStateNames.has(name));
|
|
2219
|
+
const missingInteractions = [...registryInteractions].filter((name) => !live.interactionNames.has(name));
|
|
2220
|
+
const undeclaredUiStates = [...live.uiStateNames].filter((name) => !registryUiStates.has(name));
|
|
2221
|
+
const undeclaredInteractions = [...live.interactionNames].filter((name) => !registryInteractions.has(name));
|
|
2222
|
+
if (missingUiStates.length > 0 ||
|
|
2223
|
+
missingInteractions.length > 0 ||
|
|
2224
|
+
undeclaredUiStates.length > 0 ||
|
|
2225
|
+
undeclaredInteractions.length > 0) {
|
|
2226
|
+
return planToolReceipt({
|
|
2227
|
+
ok: false,
|
|
2228
|
+
kind: "finalize_plan",
|
|
2229
|
+
error: `finalize_plan UX registry mismatch: every registered name must have one live state-flow entry and every live entry must be registered (missing uiStates: ${missingUiStates.join(", ") || "none"}; missing interactions: ${missingInteractions.join(", ") || "none"}; undeclared uiStates: ${undeclaredUiStates.join(", ") || "none"}; undeclared interactions: ${undeclaredInteractions.join(", ") || "none"})`,
|
|
2230
|
+
});
|
|
2231
|
+
}
|
|
2232
|
+
}
|
|
1616
2233
|
// Front-load the node's compile + policy gates into the finalize
|
|
1617
2234
|
// receipt (same pipeline the design-policy shell and the node
|
|
1618
2235
|
// self-check run: merge the patch onto the runtime skeleton,
|
|
@@ -1639,9 +2256,10 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1639
2256
|
catch (error) {
|
|
1640
2257
|
if (error instanceof PlanPolicyPrecheckFailure) {
|
|
1641
2258
|
// Template the fix: every uncovered interaction / state
|
|
1642
|
-
// maps to a
|
|
1643
|
-
//
|
|
1644
|
-
//
|
|
2259
|
+
// maps to a record_component_choice skeleton. Reuse is
|
|
2260
|
+
// never implied: the operator/model must fill an existing
|
|
2261
|
+
// evidencePath or switch the choice to decision=new with
|
|
2262
|
+
// the required frozen source citation.
|
|
1645
2263
|
const suggestions = error.findings
|
|
1646
2264
|
.filter((finding) => finding.code === "ui-design-coverage-missing" &&
|
|
1647
2265
|
finding.path)
|
|
@@ -1652,6 +2270,7 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1652
2270
|
purpose: finding.path,
|
|
1653
2271
|
component: "<name the existing or new component>",
|
|
1654
2272
|
decision: "reuse-existing",
|
|
2273
|
+
evidencePath: "<existing repo file that proves this reuse>",
|
|
1655
2274
|
},
|
|
1656
2275
|
},
|
|
1657
2276
|
}));
|
|
@@ -1737,6 +2356,7 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1737
2356
|
customTools: [
|
|
1738
2357
|
recordRouteSelectionTool,
|
|
1739
2358
|
recordComponentChoiceTool,
|
|
2359
|
+
recordStateRegistryTool,
|
|
1740
2360
|
recordStateFlowTool,
|
|
1741
2361
|
recordDataFlowTool,
|
|
1742
2362
|
recordMockApiTool,
|
|
@@ -1748,6 +2368,66 @@ export async function createFrontendPlanLedgerTools(input) {
|
|
|
1748
2368
|
adoptStagedFactTool,
|
|
1749
2369
|
finalizePlanTool,
|
|
1750
2370
|
],
|
|
2371
|
+
adoptCommittedFacts: async (records) => {
|
|
2372
|
+
for (const record of records) {
|
|
2373
|
+
if (record.phase !== "committed")
|
|
2374
|
+
continue;
|
|
2375
|
+
const fact = record.fact;
|
|
2376
|
+
if (!fact || typeof fact !== "object" || Array.isArray(fact))
|
|
2377
|
+
continue;
|
|
2378
|
+
const kind = typeof fact.kind === "string"
|
|
2379
|
+
? fact.kind
|
|
2380
|
+
: "plan-fact";
|
|
2381
|
+
const entry = fact.entry;
|
|
2382
|
+
const identity = (kind === "plan-requirement" || kind === "plan-verification-target") &&
|
|
2383
|
+
typeof entry?.id === "string"
|
|
2384
|
+
? `${kind}:${entry.id}`
|
|
2385
|
+
: undefined;
|
|
2386
|
+
const sourceShardPrefix = `${attemptId}:parallel-merge:${record.attemptId}:`;
|
|
2387
|
+
const requestId = `${sourceShardPrefix}${record.eventId}`;
|
|
2388
|
+
const committed = readCommittedEvents(store, attemptId);
|
|
2389
|
+
const replay = committed.find((candidate) => candidate.requestId === requestId);
|
|
2390
|
+
if (replay) {
|
|
2391
|
+
if (replay.payloadSha256 !== record.payloadSha256) {
|
|
2392
|
+
throw new Error(`frontend plan shard fact merge replay conflict: source event ${record.eventId} changed payload`);
|
|
2393
|
+
}
|
|
2394
|
+
continue;
|
|
2395
|
+
}
|
|
2396
|
+
const explicitReplacement = typeof fact.replaces === "string" &&
|
|
2397
|
+
fact.replaces === entry?.id;
|
|
2398
|
+
const conflicting = committed.find((candidate) => {
|
|
2399
|
+
const candidateFact = candidate.fact;
|
|
2400
|
+
const candidateEntry = candidateFact.entry;
|
|
2401
|
+
const sameSourceShard = candidate.requestId.startsWith(sourceShardPrefix);
|
|
2402
|
+
return (candidateFact.kind === kind &&
|
|
2403
|
+
typeof candidateEntry?.id === "string" &&
|
|
2404
|
+
`${kind}:${candidateEntry.id}` === identity &&
|
|
2405
|
+
!(sameSourceShard && explicitReplacement));
|
|
2406
|
+
});
|
|
2407
|
+
if (conflicting) {
|
|
2408
|
+
// Two shards may honestly emit the same fact (e.g. both
|
|
2409
|
+
// derive the same frozen verification target). Identical
|
|
2410
|
+
// payloads are duplicates to skip; divergent payloads are
|
|
2411
|
+
// a real conflict the ladder must resolve.
|
|
2412
|
+
if (conflicting.payloadSha256 === record.payloadSha256) {
|
|
2413
|
+
continue;
|
|
2414
|
+
}
|
|
2415
|
+
const fromSameSourceShard = conflicting.requestId.startsWith(sourceShardPrefix);
|
|
2416
|
+
if (fromSameSourceShard) {
|
|
2417
|
+
throw new Error(`frontend plan shard fact merge conflict: ${identity} was already committed by another shard with a different payload`);
|
|
2418
|
+
}
|
|
2419
|
+
// Divergent payload from a different source attempt means a
|
|
2420
|
+
// ladder retry re-derived the coverage phase: the fresh
|
|
2421
|
+
// derivation supersedes the stored record.
|
|
2422
|
+
const storeWithRecords = store;
|
|
2423
|
+
storeWithRecords.records = storeWithRecords.records.filter((candidate) => candidate.eventId !== conflicting.eventId);
|
|
2424
|
+
}
|
|
2425
|
+
const result = await adoptPlanFact(kind, requestId, fact);
|
|
2426
|
+
if (!result.ok) {
|
|
2427
|
+
throw new Error(`frontend plan shard fact merge failed: ${result.error}`);
|
|
2428
|
+
}
|
|
2429
|
+
}
|
|
2430
|
+
},
|
|
1751
2431
|
setActiveRequirementScope: (requirementIds) => {
|
|
1752
2432
|
activeRequirementScope = [
|
|
1753
2433
|
...new Set(requirementIds.filter((id) => id.trim().length > 0)),
|
|
@@ -1786,6 +2466,10 @@ function contractFactSourceSpan(fact) {
|
|
|
1786
2466
|
if (Array.isArray(fact.sourceRefs) && fact.sourceRefs.length > 0) {
|
|
1787
2467
|
return fact.sourceRefs;
|
|
1788
2468
|
}
|
|
2469
|
+
if (Array.isArray(fact.sourceFragmentIds) &&
|
|
2470
|
+
fact.sourceFragmentIds.some((value) => typeof value === "string" && value.trim().length > 0)) {
|
|
2471
|
+
return fact.sourceFragmentIds;
|
|
2472
|
+
}
|
|
1789
2473
|
if (typeof fact.source === "string" && fact.source.trim().length > 0) {
|
|
1790
2474
|
return fact.source;
|
|
1791
2475
|
}
|
|
@@ -1835,6 +2519,7 @@ export function buildFrontendTaskContractVNext(records) {
|
|
|
1835
2519
|
requirements,
|
|
1836
2520
|
constraints: byKind("constraint"),
|
|
1837
2521
|
evidenceExpectations: byKind("evidence-expectation"),
|
|
2522
|
+
deliverableDeclarations: byKind("required-deliverables").at(-1)?.items ?? [],
|
|
1838
2523
|
handoffIntents: byKind("handoff-intent"),
|
|
1839
2524
|
openQuestions: byKind("open-question"),
|
|
1840
2525
|
splitProposals: byKind("split-proposal"),
|
|
@@ -1842,7 +2527,7 @@ export function buildFrontendTaskContractVNext(records) {
|
|
|
1842
2527
|
};
|
|
1843
2528
|
}
|
|
1844
2529
|
/**
|
|
1845
|
-
* A+B: `frontend-contract-pi` incremental contract tools.
|
|
2530
|
+
* A+B: `frontend-contract-pi` incremental contract tools. The `record_*` tools
|
|
1846
2531
|
* commit origin=contract facts and `finalize_contract` commits the terminal
|
|
1847
2532
|
* disposition (ready | ready-with-assumptions | blocked + blockingOwner).
|
|
1848
2533
|
*/
|
|
@@ -1894,9 +2579,7 @@ export async function createFrontendContractTools(input) {
|
|
|
1894
2579
|
}
|
|
1895
2580
|
}
|
|
1896
2581
|
const recordKinds = {
|
|
1897
|
-
record_requirement: "requirement",
|
|
1898
2582
|
record_constraint: "constraint",
|
|
1899
|
-
record_evidence_expectation: "evidence-expectation",
|
|
1900
2583
|
record_handoff_intent: "handoff-intent",
|
|
1901
2584
|
record_open_question: "open-question",
|
|
1902
2585
|
record_split_proposal: "split-proposal",
|
|
@@ -1916,42 +2599,177 @@ export async function createFrontendContractTools(input) {
|
|
|
1916
2599
|
return receipt(result);
|
|
1917
2600
|
},
|
|
1918
2601
|
}));
|
|
1919
|
-
//
|
|
1920
|
-
//
|
|
1921
|
-
//
|
|
1922
|
-
//
|
|
1923
|
-
//
|
|
1924
|
-
//
|
|
1925
|
-
const
|
|
1926
|
-
name: "
|
|
1927
|
-
label: "
|
|
1928
|
-
description:
|
|
1929
|
-
promptSnippet: "
|
|
1930
|
-
parameters: Type.Object({
|
|
1931
|
-
path: Type.String({
|
|
1932
|
-
description: "Repo-relative candidate spec path, e.g. openspec/project-specs/ui/ucp-components-md/AdvancedSearch.md",
|
|
1933
|
-
}),
|
|
1934
|
-
disposition: Type.Enum({
|
|
1935
|
-
required: "required",
|
|
1936
|
-
relevant: "relevant",
|
|
1937
|
-
}),
|
|
1938
|
-
rationale: Type.String({}),
|
|
1939
|
-
}, { additionalProperties: false }),
|
|
2602
|
+
// record_requirement is runtime-owned by design: the requirement TEXT and
|
|
2603
|
+
// sourceFragmentIds come from the frozen source-fidelity ledger, never from
|
|
2604
|
+
// model-authored prose. The model only names the canonical id it confirms.
|
|
2605
|
+
// This closes the free-shape hole (additionalProperties:true accepted
|
|
2606
|
+
// `statement` rewrites and JSON-stringified `sourceFragmentIds` arrays,
|
|
2607
|
+
// which then reached the planner as empty text + dead fragment bindings).
|
|
2608
|
+
const recordRequirementTool = defineTool({
|
|
2609
|
+
name: "record_requirement",
|
|
2610
|
+
label: "record_requirement",
|
|
2611
|
+
description: 'Confirm one canonical ledger requirement (origin=contract requirement fact). Pass ONLY the canonical id listed in the <frontend_contract_input> inventory, e.g. {"id": "AC-001"} — the runtime commits the authoritative text and sourceFragmentIds from the frozen ledger. Never pass text/statement/sourceFragmentIds yourself: free-form rewrites and JSON-stringified fragment arrays are rejected. IMPORTANT: batch up to 5 record_* calls per message, starting from the FIRST message.',
|
|
2612
|
+
promptSnippet: "Confirm 1-5 canonical requirements by id (up to 5 per message).",
|
|
2613
|
+
parameters: Type.Object({ id: Type.String({ description: "Canonical ledger requirement id (e.g. AC-001)" }) }, { additionalProperties: false }),
|
|
1940
2614
|
async execute(_toolCallId, params) {
|
|
1941
|
-
const
|
|
1942
|
-
|
|
1943
|
-
const rationale = typeof params?.rationale === "string" ? params.rationale : "";
|
|
1944
|
-
if (!path || !disposition || !rationale.trim()) {
|
|
2615
|
+
const id = typeof params?.id === "string" ? params.id.trim() : "";
|
|
2616
|
+
if (!id) {
|
|
1945
2617
|
return receipt({
|
|
1946
2618
|
ok: false,
|
|
1947
|
-
kind: "
|
|
1948
|
-
error: "
|
|
2619
|
+
kind: "requirement",
|
|
2620
|
+
error: "record_requirement requires the canonical requirement id",
|
|
1949
2621
|
});
|
|
1950
2622
|
}
|
|
1951
|
-
const
|
|
1952
|
-
|
|
2623
|
+
const canonical = input.canonicalRequirements?.get(id);
|
|
2624
|
+
if (!canonical) {
|
|
2625
|
+
const known = [...(input.canonicalRequirements?.keys() ?? [])];
|
|
2626
|
+
return receipt({
|
|
2627
|
+
ok: false,
|
|
2628
|
+
kind: "requirement",
|
|
2629
|
+
error: `record_requirement id "${id}" is not a canonical ledger requirement; canonical ids are: ${known.join(", ") || "(none)"}`,
|
|
2630
|
+
});
|
|
2631
|
+
}
|
|
2632
|
+
const result = await adoptContractFact("requirement", {
|
|
2633
|
+
kind: "requirement",
|
|
1953
2634
|
origin: "contract",
|
|
1954
|
-
|
|
2635
|
+
disposition: "explicit",
|
|
2636
|
+
id,
|
|
2637
|
+
text: canonical.text,
|
|
2638
|
+
sourceFragmentIds: canonical.sourceFragmentIds,
|
|
2639
|
+
});
|
|
2640
|
+
return receipt(result);
|
|
2641
|
+
},
|
|
2642
|
+
});
|
|
2643
|
+
// Authoritative UI state declarations: the contract node extracts the
|
|
2644
|
+
// PRD/reference UI-state table into structured facts so the planner binds
|
|
2645
|
+
// uiStates to declared ids instead of inventing names (dogfood
|
|
2646
|
+
// dag-1788504923861-0f7b17a9: 10 invented list-visibility variants, 0 of
|
|
2647
|
+
// the 7 PRD states).
|
|
2648
|
+
const recordUiStateTool = defineTool({
|
|
2649
|
+
name: "record_ui_state",
|
|
2650
|
+
label: "record_ui_state",
|
|
2651
|
+
description: 'Declare ONE authoritative UI state extracted from the task source\'s UI-state table (origin=contract ui-state-declaration fact). Call once per declared state, exactly as the source names it: {"id": "<state id from the source table>", "trigger": "<when this state applies>", "observableOutcome": "<what the user can observe>"}. The planner must bind these ids later; do not rename or invent states.',
|
|
2652
|
+
promptSnippet: "Declare 1-5 authoritative UI states (up to 5 per message).",
|
|
2653
|
+
parameters: Type.Object({
|
|
2654
|
+
id: Type.String({ description: "UI state id exactly as the source table declares it" }),
|
|
2655
|
+
trigger: Type.String({ description: "When this state applies" }),
|
|
2656
|
+
observableOutcome: Type.String({ description: "Observable result for the user" }),
|
|
2657
|
+
}, { additionalProperties: false }),
|
|
2658
|
+
async execute(_toolCallId, params) {
|
|
2659
|
+
const id = typeof params?.id === "string" ? params.id.trim() : "";
|
|
2660
|
+
const trigger = typeof params?.trigger === "string" ? params.trigger.trim() : "";
|
|
2661
|
+
const observableOutcome = typeof params?.observableOutcome === "string"
|
|
2662
|
+
? params.observableOutcome.trim()
|
|
2663
|
+
: "";
|
|
2664
|
+
if (!id || !trigger || !observableOutcome) {
|
|
2665
|
+
return receipt({
|
|
2666
|
+
ok: false,
|
|
2667
|
+
kind: "ui-state-declaration",
|
|
2668
|
+
error: "record_ui_state requires non-empty id, trigger, and observableOutcome",
|
|
2669
|
+
});
|
|
2670
|
+
}
|
|
2671
|
+
const result = await adoptContractFact("ui-state-declaration", {
|
|
2672
|
+
kind: "ui-state-declaration",
|
|
2673
|
+
origin: "contract",
|
|
2674
|
+
id,
|
|
2675
|
+
trigger,
|
|
2676
|
+
observableOutcome,
|
|
2677
|
+
});
|
|
2678
|
+
return receipt(result);
|
|
2679
|
+
},
|
|
2680
|
+
});
|
|
2681
|
+
const evidenceStatus = Type.Enum({
|
|
2682
|
+
required: "required", optional: "optional", "not-applicable": "not-applicable",
|
|
2683
|
+
});
|
|
2684
|
+
const recordEvidenceExpectationTool = defineTool({
|
|
2685
|
+
name: "record_evidence_expectation",
|
|
2686
|
+
label: "record_evidence_expectation",
|
|
2687
|
+
description: 'Record evidence requirements as {requirementId:"<canonical id>",evidence:{static:"required|optional|not-applicable",behavior:"required|optional|not-applicable",mock:"required|optional|not-applicable","real-integration":"required|optional|not-applicable"}}. Declare at least one lane; later calls replace only the lanes they name.',
|
|
2688
|
+
parameters: Type.Object({
|
|
2689
|
+
requirementId: Type.String(),
|
|
2690
|
+
evidence: Type.Object({
|
|
2691
|
+
static: Type.Optional(evidenceStatus),
|
|
2692
|
+
behavior: Type.Optional(evidenceStatus),
|
|
2693
|
+
mock: Type.Optional(evidenceStatus),
|
|
2694
|
+
"real-integration": Type.Optional(evidenceStatus),
|
|
2695
|
+
}, { additionalProperties: false }),
|
|
2696
|
+
}, { additionalProperties: false }),
|
|
2697
|
+
async execute(_toolCallId, params) {
|
|
2698
|
+
try {
|
|
2699
|
+
const fact = frontendEvidenceExpectationSchema.parse(params);
|
|
2700
|
+
if (!input.canonicalRequirements?.has(fact.requirementId)) {
|
|
2701
|
+
throw new Error(`unknown canonical requirement ${fact.requirementId}`);
|
|
2702
|
+
}
|
|
2703
|
+
return receipt(await adoptContractFact("evidence-expectation", {
|
|
2704
|
+
kind: "evidence-expectation", origin: "contract", ...fact,
|
|
2705
|
+
}));
|
|
2706
|
+
}
|
|
2707
|
+
catch (error) {
|
|
2708
|
+
return receipt({
|
|
2709
|
+
ok: false, kind: "evidence-expectation",
|
|
2710
|
+
error: error instanceof Error ? error.message : String(error),
|
|
2711
|
+
});
|
|
2712
|
+
}
|
|
2713
|
+
},
|
|
2714
|
+
});
|
|
2715
|
+
const recordRequiredDeliverablesTool = defineTool({
|
|
2716
|
+
name: "record_required_deliverables",
|
|
2717
|
+
label: "record_required_deliverables",
|
|
2718
|
+
description: 'Declare the complete source-required file deliverables as {items:[{path,requirementId,sourceFragmentId}]}. Interpret obligations from the original source, including lists/tables: permissions (allowedPaths/only allowed to modify), prohibitions, examples and read-only references are NOT delivery obligations. Each path must appear exactly in its frozen requirement-bound source fragment. Submit {items:[]} explicitly if no files are mandatory. A correction replaces the whole inventory. Required before finalize_contract ready.',
|
|
2719
|
+
parameters: Type.Object({ items: Type.Array(Type.Object({
|
|
2720
|
+
path: Type.String(), requirementId: Type.String(), sourceFragmentId: Type.String(),
|
|
2721
|
+
}, { additionalProperties: false })) }, { additionalProperties: false }),
|
|
2722
|
+
async execute(_toolCallId, params) {
|
|
2723
|
+
try {
|
|
2724
|
+
const declaration = validateFrontendRequiredDeliverables(params, input.canonicalRequirements ?? new Map());
|
|
2725
|
+
return receipt(await adoptContractFact("required-deliverables", {
|
|
2726
|
+
kind: "required-deliverables", origin: "contract", ...declaration,
|
|
2727
|
+
}));
|
|
2728
|
+
}
|
|
2729
|
+
catch (error) {
|
|
2730
|
+
return receipt({
|
|
2731
|
+
ok: false, kind: "required-deliverables",
|
|
2732
|
+
error: error instanceof Error ? error.message : String(error),
|
|
2733
|
+
});
|
|
2734
|
+
}
|
|
2735
|
+
},
|
|
2736
|
+
});
|
|
2737
|
+
// OpenSpec selection committed as individual typed facts (one path per
|
|
2738
|
+
// call) so a large candidate set never exceeds a single model output
|
|
2739
|
+
// budget: each tool call carries exactly one {path, disposition,
|
|
2740
|
+
// rationale} row and the ledger accumulates them across calls. Only
|
|
2741
|
+
// positive classifications (required | relevant) are legal; unmentioned
|
|
2742
|
+
// candidates default to irrelevant at the prewrite gate.
|
|
2743
|
+
const recordOpenspecSelectionTool = defineTool({
|
|
2744
|
+
name: "record_openspec_selection",
|
|
2745
|
+
label: "record_openspec_selection",
|
|
2746
|
+
description: "Commit one OpenSpec candidate classification (origin=contract openspec-selection fact). Call once per path you actually use or consult: required (must be read and cited) or relevant (informs planning). Never call it for irrelevant candidates — unmentioned candidates default to irrelevant. You may call it many times; one row per call.",
|
|
2747
|
+
promptSnippet: "Commit one OpenSpec candidate classification (required | relevant); one path per call; skip irrelevant candidates.",
|
|
2748
|
+
parameters: Type.Object({
|
|
2749
|
+
path: Type.String({
|
|
2750
|
+
description: "Repo-relative candidate spec path, e.g. openspec/project-specs/ui/ucp-components-md/AdvancedSearch.md",
|
|
2751
|
+
}),
|
|
2752
|
+
disposition: Type.Enum({
|
|
2753
|
+
required: "required",
|
|
2754
|
+
relevant: "relevant",
|
|
2755
|
+
}),
|
|
2756
|
+
rationale: Type.String({}),
|
|
2757
|
+
}, { additionalProperties: false }),
|
|
2758
|
+
async execute(_toolCallId, params) {
|
|
2759
|
+
const path = typeof params?.path === "string" ? params.path : "";
|
|
2760
|
+
const disposition = params?.disposition;
|
|
2761
|
+
const rationale = typeof params?.rationale === "string" ? params.rationale : "";
|
|
2762
|
+
if (!path || !disposition || !rationale.trim()) {
|
|
2763
|
+
return receipt({
|
|
2764
|
+
ok: false,
|
|
2765
|
+
kind: "openspec-selection",
|
|
2766
|
+
error: "record_openspec_selection requires non-empty path, disposition (required|relevant), and rationale",
|
|
2767
|
+
});
|
|
2768
|
+
}
|
|
2769
|
+
const result = await adoptContractFact("openspec-selection", {
|
|
2770
|
+
kind: "openspec-selection",
|
|
2771
|
+
origin: "contract",
|
|
2772
|
+
path,
|
|
1955
2773
|
disposition,
|
|
1956
2774
|
rationale,
|
|
1957
2775
|
});
|
|
@@ -1986,6 +2804,13 @@ export async function createFrontendContractTools(input) {
|
|
|
1986
2804
|
error: "blocked disposition requires blockingOwner (blocked-human | blocked-external)",
|
|
1987
2805
|
});
|
|
1988
2806
|
}
|
|
2807
|
+
if (disposition !== "blocked" && input.canonicalRequirements?.size &&
|
|
2808
|
+
!readCommittedEvents(store, attemptId).some((record) => record.fact.kind === "required-deliverables")) {
|
|
2809
|
+
return receipt({
|
|
2810
|
+
ok: false, kind: "finalize_contract",
|
|
2811
|
+
error: "call record_required_deliverables with the complete source-bound inventory (or items:[] when none) before finalizing",
|
|
2812
|
+
});
|
|
2813
|
+
}
|
|
1989
2814
|
const blockedOwner = mapContractBlockedOwner({
|
|
1990
2815
|
disposition: disposition ?? "",
|
|
1991
2816
|
blockingOwner,
|
|
@@ -2009,6 +2834,10 @@ export async function createFrontendContractTools(input) {
|
|
|
2009
2834
|
return {
|
|
2010
2835
|
customTools: [
|
|
2011
2836
|
...recordTools,
|
|
2837
|
+
recordRequirementTool,
|
|
2838
|
+
recordEvidenceExpectationTool,
|
|
2839
|
+
recordUiStateTool,
|
|
2840
|
+
recordRequiredDeliverablesTool,
|
|
2012
2841
|
recordOpenspecSelectionTool,
|
|
2013
2842
|
finalizeContractTool,
|
|
2014
2843
|
],
|
|
@@ -2228,6 +3057,22 @@ export async function createFrontendScoutEvidenceTools(input) {
|
|
|
2228
3057
|
});
|
|
2229
3058
|
return {
|
|
2230
3059
|
customTools: [recordTargetSurfaceTool, recordDesignEvidenceTool],
|
|
3060
|
+
committedFacts: () => readCommittedEvents(store, attemptId),
|
|
3061
|
+
adoptCommittedFacts: async (records) => {
|
|
3062
|
+
for (const record of records) {
|
|
3063
|
+
if (record.phase !== "committed")
|
|
3064
|
+
continue;
|
|
3065
|
+
const fact = record.fact;
|
|
3066
|
+
if (!fact || typeof fact !== "object" || Array.isArray(fact))
|
|
3067
|
+
continue;
|
|
3068
|
+
const result = await adoptScoutFact(typeof fact.kind === "string"
|
|
3069
|
+
? fact.kind
|
|
3070
|
+
: "scout-fact", fact);
|
|
3071
|
+
if (!result.ok) {
|
|
3072
|
+
throw new Error(`frontend scout shard fact merge failed: ${result.error}`);
|
|
3073
|
+
}
|
|
3074
|
+
}
|
|
3075
|
+
},
|
|
2231
3076
|
flush: async () => {
|
|
2232
3077
|
const committed = readCommittedEvents(store, attemptId);
|
|
2233
3078
|
await writeTypedEventStoreJsonl(path.join(input.runDir, input.nodeId, "scout-typed-facts.jsonl"), committed);
|
|
@@ -2620,19 +3465,30 @@ const FRONTEND_PLAN_SEGMENTS = [
|
|
|
2620
3465
|
instruction: [
|
|
2621
3466
|
"PLAN PHASE — requirement coverage only.",
|
|
2622
3467
|
"Your ONLY job: for every frozen requirement, emit record_plan_requirement (requirement → implementation files) and record_plan_verification_target facts (verification target bound to requirement ids and files). Group related requirements under one non-static behavior target when one observable test behavior proves them together; do not mechanically create one target per requirement. A non-static target id is the stable machine trace token; never submit prose as a symbol. Do NOT record components, UI states, mock, dependency, or routes — a follow-up session owns those.",
|
|
3468
|
+
"A coverage session is complete only when EVERY requirement assigned to this session (the full inventory, or the exact COVERAGE BATCH / shard list when present) has committed coverage facts: a record_plan_requirement entry plus verification targets, or a committed evidence gap. Keep committing in batches of up to 4 record_* calls per assistant message until then; do not write a concluding summary while any assigned requirement is still uncommitted — an early stop strands the remainder into a MISSING-FACT repair session and doubles the sessions needed.",
|
|
2623
3469
|
"If a requirement genuinely cannot have a verification target, record a non-empty record_plan_evidence_gap. Do not call finalize_plan; it is not available in this phase.",
|
|
2624
3470
|
].join(" "),
|
|
2625
3471
|
},
|
|
3472
|
+
{
|
|
3473
|
+
id: "ux-registry",
|
|
3474
|
+
toolNames: new Set(["record_state_registry", "adopt_staged_fact"]),
|
|
3475
|
+
instruction: [
|
|
3476
|
+
"PLAN PHASE — global UX vocabulary.",
|
|
3477
|
+
"Review ALL frozen requirements together and call record_state_registry exactly once with the complete UI-state and interaction vocabulary. UI states use declaredUiStates ids when present. Interaction names are stable kebab-case behavior domains; merge requirements that describe the same behavior instead of renaming it per AC slice. Empty arrays explicitly declare that no UX vocabulary applies. Do not record component choices or state-flow details in this phase.",
|
|
3478
|
+
"Do not call finalize_plan; it is not available in this phase.",
|
|
3479
|
+
].join(" "),
|
|
3480
|
+
},
|
|
2626
3481
|
{
|
|
2627
3482
|
id: "ux-local",
|
|
2628
3483
|
toolNames: new Set([
|
|
2629
3484
|
"record_component_choice",
|
|
2630
3485
|
"record_state_flow",
|
|
3486
|
+
"record_plan_verification_target",
|
|
2631
3487
|
"adopt_staged_fact",
|
|
2632
3488
|
]),
|
|
2633
3489
|
instruction: [
|
|
2634
|
-
"PLAN PHASE —
|
|
2635
|
-
"Requirements and verification targets are already committed in the ledger
|
|
3490
|
+
"PLAN PHASE — global UX decisions.",
|
|
3491
|
+
"Requirements and verification targets are already committed in the ledger. Review the complete requirement set and the committed global UX registry together, then record each component choice, UI state and interaction. Bind each applicable state to its verificationTargetIds; the runtime derives the reverse VT.uiStates relation. If a VT requires correction, record_plan_verification_target with replace:true is available after declaring its states; preserve its requirement coverage. Multiple requirements describing one behavior share one registry name and state-flow entry. Cross-cutting data flow belongs to the global Mock/data phase. Do not record routes, Mock/API policy, dependencies, or design deviations here.",
|
|
2636
3492
|
"Do not call finalize_plan; it is not available in this phase.",
|
|
2637
3493
|
].join(" "),
|
|
2638
3494
|
},
|
|
@@ -2690,40 +3546,6 @@ function planFactStringList(value) {
|
|
|
2690
3546
|
function planFactScopeIntersects(fact, requirementIds) {
|
|
2691
3547
|
return planFactStringList(fact.scopeRequirementIds).some((id) => requirementIds.has(id));
|
|
2692
3548
|
}
|
|
2693
|
-
/** Apply state-flow replacement/removal semantics exactly as the plan compiler does. */
|
|
2694
|
-
function collectCanonicalStateFlowNames(committedFacts) {
|
|
2695
|
-
const uiStateNames = new Set();
|
|
2696
|
-
const interactionNames = new Set();
|
|
2697
|
-
for (const value of committedFacts) {
|
|
2698
|
-
const fact = committedFactFromPlanRecord(value);
|
|
2699
|
-
if (!fact || fact.origin !== "plan" || fact.kind !== "state-flow")
|
|
2700
|
-
continue;
|
|
2701
|
-
for (const name of planFactStringList(fact.removeUiStateNames)) {
|
|
2702
|
-
uiStateNames.delete(name);
|
|
2703
|
-
}
|
|
2704
|
-
for (const name of planFactStringList(fact.removeInteractionNames)) {
|
|
2705
|
-
interactionNames.delete(name);
|
|
2706
|
-
}
|
|
2707
|
-
for (const state of Array.isArray(fact.uiStates) ? fact.uiStates : []) {
|
|
2708
|
-
if (!state || typeof state !== "object" || Array.isArray(state))
|
|
2709
|
-
continue;
|
|
2710
|
-
const name = state.name;
|
|
2711
|
-
if (typeof name === "string" && name.trim())
|
|
2712
|
-
uiStateNames.add(name);
|
|
2713
|
-
}
|
|
2714
|
-
for (const interaction of Array.isArray(fact.interactions)
|
|
2715
|
-
? fact.interactions
|
|
2716
|
-
: []) {
|
|
2717
|
-
if (!interaction || typeof interaction !== "object" || Array.isArray(interaction)) {
|
|
2718
|
-
continue;
|
|
2719
|
-
}
|
|
2720
|
-
const name = interaction.name;
|
|
2721
|
-
if (typeof name === "string" && name.trim())
|
|
2722
|
-
interactionNames.add(name);
|
|
2723
|
-
}
|
|
2724
|
-
}
|
|
2725
|
-
return { uiStateNames, interactionNames };
|
|
2726
|
-
}
|
|
2727
3549
|
/** Compute the authoritative coverage queue from the committed plan ledger. */
|
|
2728
3550
|
export function collectFrontendPlanMissingFacts(input) {
|
|
2729
3551
|
const requirements = new Map();
|
|
@@ -2805,6 +3627,17 @@ export function collectFrontendPlanPhaseMissingFacts(input) {
|
|
|
2805
3627
|
const facts = input.committedFacts
|
|
2806
3628
|
.map(committedFactFromPlanRecord)
|
|
2807
3629
|
.filter((fact) => Boolean(fact && fact.origin === "plan"));
|
|
3630
|
+
if (input.phase === "ux-registry") {
|
|
3631
|
+
return facts.some((fact) => fact.kind === "state-registry")
|
|
3632
|
+
? []
|
|
3633
|
+
: [
|
|
3634
|
+
{
|
|
3635
|
+
kind: "state-registry",
|
|
3636
|
+
requirementIds: [...input.requirementIds],
|
|
3637
|
+
reason: "global UX vocabulary phase has no committed state-registry fact",
|
|
3638
|
+
},
|
|
3639
|
+
];
|
|
3640
|
+
}
|
|
2808
3641
|
if (input.phase === "ux-local") {
|
|
2809
3642
|
const needsUx = input.requirementIds.some((id) => input.behaviorRequiredRequirementIds?.includes(id));
|
|
2810
3643
|
if (!needsUx)
|
|
@@ -2899,11 +3732,28 @@ export function batchFrontendPlanRequirements(input) {
|
|
|
2899
3732
|
return batches;
|
|
2900
3733
|
}
|
|
2901
3734
|
const FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS = 10;
|
|
2902
|
-
const
|
|
2903
|
-
// A large requirement set creates
|
|
2904
|
-
//
|
|
2905
|
-
//
|
|
3735
|
+
const FRONTEND_PLAN_COVERAGE_MAX_CONCURRENCY = 4;
|
|
3736
|
+
// A large requirement set creates parallel coverage shards; UX remains one
|
|
3737
|
+
// global decision session so behavior names and component/state facts are not
|
|
3738
|
+
// reinvented per AC batch. Keep a safety bound for adaptive coverage retries
|
|
3739
|
+
// without letting the old 32-session ceiling skip finalize.
|
|
2906
3740
|
const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 128;
|
|
3741
|
+
const FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS = 8;
|
|
3742
|
+
const FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS = 24;
|
|
3743
|
+
async function mapWithConcurrency(items, limit, worker) {
|
|
3744
|
+
const results = new Array(items.length);
|
|
3745
|
+
let nextIndex = 0;
|
|
3746
|
+
const workerCount = Math.min(Math.max(1, limit), items.length);
|
|
3747
|
+
await Promise.all(Array.from({ length: workerCount }, async () => {
|
|
3748
|
+
while (true) {
|
|
3749
|
+
const index = nextIndex++;
|
|
3750
|
+
if (index >= items.length)
|
|
3751
|
+
return;
|
|
3752
|
+
results[index] = await worker(items[index], index);
|
|
3753
|
+
}
|
|
3754
|
+
}));
|
|
3755
|
+
return results;
|
|
3756
|
+
}
|
|
2907
3757
|
function compactPromptString(value, maxChars) {
|
|
2908
3758
|
if (typeof value !== "string" || value.trim().length === 0)
|
|
2909
3759
|
return undefined;
|
|
@@ -2920,6 +3770,45 @@ function compactPromptStringArray(value, maxEntries = 12, maxChars = 180) {
|
|
|
2920
3770
|
.filter((item) => item !== undefined)
|
|
2921
3771
|
.slice(0, maxEntries);
|
|
2922
3772
|
}
|
|
3773
|
+
function countFrontendPlanTargetSurfaces(basePrompt) {
|
|
3774
|
+
const match = /<frontend_plan_input>[\s\S]*?<\/frontend_plan_input>/.exec(basePrompt);
|
|
3775
|
+
if (!match)
|
|
3776
|
+
return undefined;
|
|
3777
|
+
for (const line of match[0].split(/\r?\n/)) {
|
|
3778
|
+
try {
|
|
3779
|
+
const payload = JSON.parse(line);
|
|
3780
|
+
if (Array.isArray(payload.targetSurface))
|
|
3781
|
+
return payload.targetSurface.length;
|
|
3782
|
+
}
|
|
3783
|
+
catch {
|
|
3784
|
+
// surrounding lines are prose
|
|
3785
|
+
}
|
|
3786
|
+
}
|
|
3787
|
+
return undefined;
|
|
3788
|
+
}
|
|
3789
|
+
/**
|
|
3790
|
+
* Deterministically extract repository file paths that frozen verification
|
|
3791
|
+
* commands operate on (`--config <file>`, `node --check <file>`). A frozen
|
|
3792
|
+
* command whose referenced file is outside the planner's verification targets
|
|
3793
|
+
* can never run: the admission writeSet derives from those targets, so the
|
|
3794
|
+
* writer is not authorized to create the file and verify-shell fails ~20
|
|
3795
|
+
* minutes later. Surfacing the gap as plan missing-facts lets the coverage or
|
|
3796
|
+
* compact session record the missing verification target inside the same
|
|
3797
|
+
* attempt instead.
|
|
3798
|
+
*/
|
|
3799
|
+
export function collectFrontendVerificationCommandFiles(basePrompt) {
|
|
3800
|
+
const files = new Set();
|
|
3801
|
+
const configRe = /--config\s+([\w@./-]+\.(?:js|mjs|cjs|ts|json))/g;
|
|
3802
|
+
const checkRe = /node\s+--check\s+([\w@./-]+\.(?:js|mjs|cjs))/g;
|
|
3803
|
+
for (const re of [configRe, checkRe]) {
|
|
3804
|
+
for (const match of basePrompt.matchAll(re)) {
|
|
3805
|
+
const file = match[1];
|
|
3806
|
+
if (file && file.includes("/"))
|
|
3807
|
+
files.add(file);
|
|
3808
|
+
}
|
|
3809
|
+
}
|
|
3810
|
+
return [...files].sort();
|
|
3811
|
+
}
|
|
2923
3812
|
/**
|
|
2924
3813
|
* Remove the repeated full planner input from a coverage batch. The normal
|
|
2925
3814
|
* plan prompt already contains a bounded JSON handoff, but repeating all
|
|
@@ -2990,8 +3879,8 @@ export function compactFrontendPlanPromptForRequirementSlice(basePrompt, require
|
|
|
2990
3879
|
...(compactPromptString(target.id, 80)
|
|
2991
3880
|
? { id: compactPromptString(target.id, 80) }
|
|
2992
3881
|
: {}),
|
|
2993
|
-
...(compactPromptString(target.
|
|
2994
|
-
? {
|
|
3882
|
+
...(compactPromptString(target.commandId, 80)
|
|
3883
|
+
? { commandId: compactPromptString(target.commandId, 80) }
|
|
2995
3884
|
: {}),
|
|
2996
3885
|
...(compactPromptString(target.commandLabel, 180)
|
|
2997
3886
|
? { commandLabel: compactPromptString(target.commandLabel, 180) }
|
|
@@ -3047,8 +3936,33 @@ export function compactFrontendPlanPromptForRequirementSlice(basePrompt, require
|
|
|
3047
3936
|
}];
|
|
3048
3937
|
})
|
|
3049
3938
|
: [];
|
|
3939
|
+
const compactDeclaredUiStates = Array.isArray(payload.declaredUiStates)
|
|
3940
|
+
? payload.declaredUiStates.flatMap((value) => {
|
|
3941
|
+
if (!value || typeof value !== "object" || Array.isArray(value))
|
|
3942
|
+
return [];
|
|
3943
|
+
const state = value;
|
|
3944
|
+
const id = compactPromptString(state.id, 80);
|
|
3945
|
+
if (!id)
|
|
3946
|
+
return [];
|
|
3947
|
+
return [{
|
|
3948
|
+
id,
|
|
3949
|
+
...(compactPromptString(state.trigger, 180)
|
|
3950
|
+
? { trigger: compactPromptString(state.trigger, 180) }
|
|
3951
|
+
: {}),
|
|
3952
|
+
...(compactPromptString(state.observableOutcome, 240)
|
|
3953
|
+
? { observableOutcome: compactPromptString(state.observableOutcome, 240) }
|
|
3954
|
+
: {}),
|
|
3955
|
+
}];
|
|
3956
|
+
})
|
|
3957
|
+
: [];
|
|
3958
|
+
const committedUx = payload.committedUx &&
|
|
3959
|
+
typeof payload.committedUx === "object" &&
|
|
3960
|
+
!Array.isArray(payload.committedUx)
|
|
3961
|
+
? payload.committedUx
|
|
3962
|
+
: undefined;
|
|
3050
3963
|
const compactPayload = {
|
|
3051
3964
|
requirements: compactRequirements,
|
|
3965
|
+
requiredDeliverables: Array.isArray(payload.requiredDeliverables) ? payload.requiredDeliverables : [],
|
|
3052
3966
|
...(compactTargetSurface.length > 0
|
|
3053
3967
|
? { targetSurface: compactTargetSurface }
|
|
3054
3968
|
: {}),
|
|
@@ -3058,6 +3972,17 @@ export function compactFrontendPlanPromptForRequirementSlice(basePrompt, require
|
|
|
3058
3972
|
...(compactDesignEvidence.length > 0
|
|
3059
3973
|
? { designEvidence: compactDesignEvidence }
|
|
3060
3974
|
: {}),
|
|
3975
|
+
...(compactDeclaredUiStates.length > 0
|
|
3976
|
+
? { declaredUiStates: compactDeclaredUiStates }
|
|
3977
|
+
: {}),
|
|
3978
|
+
...(committedUx
|
|
3979
|
+
? {
|
|
3980
|
+
committedUx: {
|
|
3981
|
+
uiStateNames: compactPromptStringArray(committedUx.uiStateNames, 40, 100),
|
|
3982
|
+
interactionNames: compactPromptStringArray(committedUx.interactionNames, 60, 120),
|
|
3983
|
+
},
|
|
3984
|
+
}
|
|
3985
|
+
: {}),
|
|
3061
3986
|
};
|
|
3062
3987
|
const compactBlock = [
|
|
3063
3988
|
"<frontend_plan_input>",
|
|
@@ -3136,8 +4061,7 @@ function compactFrontendPlanLedgerContext(input) {
|
|
|
3136
4061
|
kind: fact.kind,
|
|
3137
4062
|
entry: {
|
|
3138
4063
|
id: compactPromptString(entry.id, 80),
|
|
3139
|
-
|
|
3140
|
-
commandLabel: compactPromptString(entry.commandLabel, 180),
|
|
4064
|
+
commandId: compactPromptString(entry.commandId, 80),
|
|
3141
4065
|
file: compactPromptString(entry.file, 180),
|
|
3142
4066
|
requirementIds,
|
|
3143
4067
|
uiStates: compactPromptStringArray(entry.uiStates, 12, 100),
|
|
@@ -3155,6 +4079,8 @@ function compactFrontendPlanLedgerContext(input) {
|
|
|
3155
4079
|
purpose: compactPromptString(item.purpose, 120),
|
|
3156
4080
|
component: compactPromptString(item.component, 120),
|
|
3157
4081
|
decision: compactPromptString(item.decision, 40),
|
|
4082
|
+
covers: compactPromptStringArray(item.covers, 40, 120),
|
|
4083
|
+
evidencePath: compactPromptString(item.evidencePath, 180),
|
|
3158
4084
|
}];
|
|
3159
4085
|
}).slice(0, 24)
|
|
3160
4086
|
: [];
|
|
@@ -3162,6 +4088,14 @@ function compactFrontendPlanLedgerContext(input) {
|
|
|
3162
4088
|
compactFacts.push({ kind: fact.kind, uiComponentChoices: choices });
|
|
3163
4089
|
continue;
|
|
3164
4090
|
}
|
|
4091
|
+
if (fact.kind === "state-registry") {
|
|
4092
|
+
compactFacts.push({
|
|
4093
|
+
kind: fact.kind,
|
|
4094
|
+
uiStateNames: compactPromptStringArray(fact.uiStateNames, 40, 100),
|
|
4095
|
+
interactionNames: compactPromptStringArray(fact.interactionNames, 60, 120),
|
|
4096
|
+
});
|
|
4097
|
+
continue;
|
|
4098
|
+
}
|
|
3165
4099
|
if (fact.kind === "state-flow") {
|
|
3166
4100
|
compactFacts.push({
|
|
3167
4101
|
kind: fact.kind,
|
|
@@ -3224,8 +4158,10 @@ function compactFrontendPlanLedgerContext(input) {
|
|
|
3224
4158
|
export async function runFrontendPlanSegmentedSessions(input) {
|
|
3225
4159
|
const queue = [];
|
|
3226
4160
|
const coverageSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "coverage");
|
|
4161
|
+
const uxRegistrySegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "ux-registry");
|
|
3227
4162
|
const uxSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "ux-local");
|
|
3228
4163
|
const globalMockDataSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "global-mock-data");
|
|
4164
|
+
const finalizeSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "finalize");
|
|
3229
4165
|
const allRequirementIds = input.requirementIds ?? [];
|
|
3230
4166
|
const buildPhasePrompt = (segment, missing = []) => {
|
|
3231
4167
|
const compact = allRequirementIds.length > 0
|
|
@@ -3242,7 +4178,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3242
4178
|
? ["plan-requirement", "plan-verification-target", "state-flow", "data-flow", "mock-api"]
|
|
3243
4179
|
: segment.id === "global-dependency-deviation"
|
|
3244
4180
|
? ["dependency", "design-deviation"]
|
|
3245
|
-
: ["plan-requirement", "plan-verification-target", "component-choice", "state-flow", "data-flow", "mock-api", "design-deviation", "dependency", "target-surface"];
|
|
4181
|
+
: ["plan-requirement", "plan-verification-target", "state-registry", "component-choice", "state-flow", "data-flow", "mock-api", "design-deviation", "dependency", "target-surface"];
|
|
3246
4182
|
const ledger = input.committedFacts
|
|
3247
4183
|
? compactFrontendPlanLedgerContext({
|
|
3248
4184
|
committedFacts: input.committedFacts(),
|
|
@@ -3265,7 +4201,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3265
4201
|
const buildCoveragePrompt = (slice, missing = []) => [
|
|
3266
4202
|
compactFrontendPlanPromptForRequirementSlice(input.basePrompt, slice),
|
|
3267
4203
|
coverageSegment.instruction,
|
|
3268
|
-
`COVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them.`,
|
|
4204
|
+
`COVERAGE BATCH: process ONLY these requirements in this session: ${slice.join(", ")}. Other requirements are handled by separate sessions; do not record them. Finish the whole list before concluding: commit every listed requirement's coverage facts (verification targets or an evidence gap), batching up to 4 record_* calls per message; an early stop re-queues the remainder as a MISSING-FACT repair session.`,
|
|
3269
4205
|
...(missing.length > 0
|
|
3270
4206
|
? [
|
|
3271
4207
|
"MISSING-FACT QUEUE: the previous session did not establish complete coverage. Repair ONLY these items, then re-check the slice:",
|
|
@@ -3278,14 +4214,20 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3278
4214
|
? compactFrontendPlanLedgerContext({
|
|
3279
4215
|
committedFacts: input.committedFacts(),
|
|
3280
4216
|
requirementIds: slice,
|
|
3281
|
-
kinds: [
|
|
4217
|
+
kinds: [
|
|
4218
|
+
"plan-requirement",
|
|
4219
|
+
"plan-verification-target",
|
|
4220
|
+
"state-registry",
|
|
4221
|
+
"component-choice",
|
|
4222
|
+
"state-flow",
|
|
4223
|
+
],
|
|
3282
4224
|
scopedKinds: ["component-choice", "state-flow"],
|
|
3283
4225
|
})
|
|
3284
4226
|
: "";
|
|
3285
4227
|
return [
|
|
3286
4228
|
compactFrontendPlanPromptForRequirementSlice(input.basePrompt, slice),
|
|
3287
4229
|
uxSegment.instruction,
|
|
3288
|
-
`UX
|
|
4230
|
+
`GLOBAL UX SCOPE: process the complete requirement set together: ${slice.join(", ")}. Do not split or rename one behavior by requirement id.`,
|
|
3289
4231
|
ledger,
|
|
3290
4232
|
...(missing.length > 0
|
|
3291
4233
|
? [
|
|
@@ -3295,6 +4237,74 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3295
4237
|
: []),
|
|
3296
4238
|
].filter(Boolean).join("\n\n");
|
|
3297
4239
|
};
|
|
4240
|
+
const buildUxRegistryPrompt = (missing = []) => {
|
|
4241
|
+
const ledger = input.committedFacts
|
|
4242
|
+
? compactFrontendPlanLedgerContext({
|
|
4243
|
+
committedFacts: input.committedFacts(),
|
|
4244
|
+
requirementIds: allRequirementIds,
|
|
4245
|
+
kinds: [
|
|
4246
|
+
"plan-requirement",
|
|
4247
|
+
"plan-verification-target",
|
|
4248
|
+
"state-registry",
|
|
4249
|
+
],
|
|
4250
|
+
})
|
|
4251
|
+
: "";
|
|
4252
|
+
const compact = allRequirementIds.length > 0
|
|
4253
|
+
? compactFrontendPlanPromptForRequirementSlice(input.basePrompt, allRequirementIds, {
|
|
4254
|
+
includeVerificationTargets: false,
|
|
4255
|
+
includeDesignEvidence: false,
|
|
4256
|
+
includeChecklist: false,
|
|
4257
|
+
})
|
|
4258
|
+
: input.basePrompt;
|
|
4259
|
+
return [
|
|
4260
|
+
compact,
|
|
4261
|
+
uxRegistrySegment.instruction,
|
|
4262
|
+
ledger,
|
|
4263
|
+
...(missing.length > 0
|
|
4264
|
+
? [
|
|
4265
|
+
"MISSING-FACT QUEUE: commit the global UX registry, then re-check this phase:",
|
|
4266
|
+
...missing.map((item) => `- ${item.reason}`),
|
|
4267
|
+
]
|
|
4268
|
+
: []),
|
|
4269
|
+
].filter(Boolean).join("\n\n");
|
|
4270
|
+
};
|
|
4271
|
+
const buildCompactLocalPrompt = (missing = []) => {
|
|
4272
|
+
const ledger = input.committedFacts
|
|
4273
|
+
? compactFrontendPlanLedgerContext({
|
|
4274
|
+
committedFacts: input.committedFacts(),
|
|
4275
|
+
requirementIds: allRequirementIds,
|
|
4276
|
+
kinds: [
|
|
4277
|
+
"plan-requirement",
|
|
4278
|
+
"plan-verification-target",
|
|
4279
|
+
"state-registry",
|
|
4280
|
+
"component-choice",
|
|
4281
|
+
"state-flow",
|
|
4282
|
+
],
|
|
4283
|
+
})
|
|
4284
|
+
: "";
|
|
4285
|
+
return [
|
|
4286
|
+
compactFrontendPlanPromptForRequirementSlice(input.basePrompt, allRequirementIds),
|
|
4287
|
+
"PLAN PHASE — compact local planning for a small frontend request.",
|
|
4288
|
+
"Review all listed requirements together and record_state_registry first with one global UX vocabulary. Then record every plan-requirement and verification-target fact, followed by the component-choice and state-flow facts needed by the observable UX. Do not read the repository or task source; use only the committed input above. Do not call finalize_plan in this session.",
|
|
4289
|
+
"TOOL-FIRST: your first assistant actions must be record_* tool calls, at most 2-3 facts per message. Do not draft the whole analysis before recording; if a fact is uncertain, record it with an evidence gap instead of reasoning longer.",
|
|
4290
|
+
ledger,
|
|
4291
|
+
...(missing.length > 0
|
|
4292
|
+
? [
|
|
4293
|
+
"MISSING-FACT QUEUE: repair ONLY these items, then re-check the compact local phase:",
|
|
4294
|
+
...missing.map((item) => `- ${item.kind}${item.id ? ` ${item.id}` : ""} for ${item.requirementIds.join(", ")}: ${item.reason}`),
|
|
4295
|
+
]
|
|
4296
|
+
: []),
|
|
4297
|
+
].filter(Boolean).join("\n\n");
|
|
4298
|
+
};
|
|
4299
|
+
const compactFinalizeInstruction = "This is a small-request compact pass. Reconcile the committed local facts with route, data-flow, Mock/API, dependency and deviation policy, then call finalize_plan exactly once.";
|
|
4300
|
+
const buildCompactFinalizePrompt = (missing = []) => [buildPhasePrompt(finalizeSegment, missing), compactFinalizeInstruction].join("\n\n");
|
|
4301
|
+
const mapPlannerExhaustion = (r, committedAnyFacts) => isPlannerThinkingExhausted(r, committedAnyFacts)
|
|
4302
|
+
? {
|
|
4303
|
+
...r,
|
|
4304
|
+
failureCategory: OUTPUT_LIMIT_RETRY_CATEGORY,
|
|
4305
|
+
stderr: `${r.stderr}\n${OUTPUT_LIMIT_RETRY_CATEGORY}: stopReason=length ended the turn before the next typed fact; preserve committed facts and retry only the unfinished phase`.trim(),
|
|
4306
|
+
}
|
|
4307
|
+
: r;
|
|
3298
4308
|
// An empty list means "ledger unreadable / unknown" and falls back to one
|
|
3299
4309
|
// unscoped coverage session; a non-empty list enables deterministic sharding.
|
|
3300
4310
|
const requirementIdsProvided = input.requirementIds !== undefined && input.requirementIds.length > 0;
|
|
@@ -3307,7 +4317,66 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3307
4317
|
: [];
|
|
3308
4318
|
const incompleteRequirementIds = new Set(initialMissing.flatMap((item) => item.requirementIds));
|
|
3309
4319
|
const coverageWorkIds = (input.requirementIds ?? []).filter((id) => pending.includes(id) || incompleteRequirementIds.has(id));
|
|
3310
|
-
if (
|
|
4320
|
+
if (input.parallelCoverageOnly === true &&
|
|
4321
|
+
requirementIdsProvided &&
|
|
4322
|
+
coverageWorkIds.length === 0) {
|
|
4323
|
+
// A retry may reopen a shard whose committed facts are already complete.
|
|
4324
|
+
// Treat that shard as an idempotent no-op; returning the normal empty
|
|
4325
|
+
// session failure would make a partially failed map impossible to resume.
|
|
4326
|
+
return {
|
|
4327
|
+
ok: true,
|
|
4328
|
+
assistantText: "",
|
|
4329
|
+
command: [],
|
|
4330
|
+
durationMs: 0,
|
|
4331
|
+
exitCode: 0,
|
|
4332
|
+
failureCategory: "success",
|
|
4333
|
+
modelDisplay: "reused-coverage-facts",
|
|
4334
|
+
parsedEvents: 0,
|
|
4335
|
+
stderr: "",
|
|
4336
|
+
stdout: "",
|
|
4337
|
+
timedOut: false,
|
|
4338
|
+
attemptedModels: [],
|
|
4339
|
+
fallbackUsed: false,
|
|
4340
|
+
tokensUsed: 0,
|
|
4341
|
+
};
|
|
4342
|
+
}
|
|
4343
|
+
const estimatedCalls = (input.requirementIds ?? []).reduce((total, id) => total + Math.max(1, input.requirementCosts?.get(id) ?? 2), 0);
|
|
4344
|
+
const targetSurfaceCount = countFrontendPlanTargetSurfaces(input.basePrompt);
|
|
4345
|
+
// Small, single-surface requests do not benefit from six isolated Pi
|
|
4346
|
+
// sessions. Keep the typed ledger as the authority, but let one local
|
|
4347
|
+
// session establish requirement/UX facts and one final session establish
|
|
4348
|
+
// cross-cutting policy + finalize. The old sharded ladder remains available
|
|
4349
|
+
// for larger plans and for the unscoped compatibility path.
|
|
4350
|
+
const useCompactSmallPlan = requirementIdsProvided &&
|
|
4351
|
+
input.compactSmallPlan === true &&
|
|
4352
|
+
input.requirementCosts !== undefined &&
|
|
4353
|
+
estimatedCalls > 12 &&
|
|
4354
|
+
(input.requirementIds?.length ?? 0) <= FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS &&
|
|
4355
|
+
estimatedCalls <= FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS &&
|
|
4356
|
+
targetSurfaceCount === 1;
|
|
4357
|
+
if (useCompactSmallPlan) {
|
|
4358
|
+
const compactLocalTools = new Set([
|
|
4359
|
+
"record_plan_requirement",
|
|
4360
|
+
"record_plan_verification_target",
|
|
4361
|
+
"record_plan_evidence_gap",
|
|
4362
|
+
"record_state_registry",
|
|
4363
|
+
"record_component_choice",
|
|
4364
|
+
"record_state_flow",
|
|
4365
|
+
"adopt_staged_fact",
|
|
4366
|
+
]);
|
|
4367
|
+
queue.push({
|
|
4368
|
+
id: "compact-local",
|
|
4369
|
+
toolNames: compactLocalTools,
|
|
4370
|
+
requirementSlice: [...allRequirementIds],
|
|
4371
|
+
prompt: buildCompactLocalPrompt(),
|
|
4372
|
+
});
|
|
4373
|
+
queue.push({
|
|
4374
|
+
id: "finalize",
|
|
4375
|
+
toolNames: null,
|
|
4376
|
+
prompt: buildCompactFinalizePrompt(),
|
|
4377
|
+
});
|
|
4378
|
+
}
|
|
4379
|
+
else if (!requirementIdsProvided) {
|
|
3311
4380
|
queue.push({
|
|
3312
4381
|
id: "coverage",
|
|
3313
4382
|
toolNames: coverageSegment.toolNames,
|
|
@@ -3329,42 +4398,47 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3329
4398
|
prompt: buildCoveragePrompt(slice),
|
|
3330
4399
|
}));
|
|
3331
4400
|
}
|
|
3332
|
-
|
|
3333
|
-
|
|
3334
|
-
|
|
3335
|
-
|
|
3336
|
-
|
|
3337
|
-
|
|
3338
|
-
|
|
3339
|
-
|
|
3340
|
-
|
|
3341
|
-
|
|
3342
|
-
}
|
|
3343
|
-
: [
|
|
3344
|
-
uxSlices.forEach((slice, batchIndex) => queue.push({
|
|
3345
|
-
id: `ux-local-${batchIndex + 1}`,
|
|
3346
|
-
toolNames: uxSegment.toolNames,
|
|
3347
|
-
...(slice.length > 0 ? { requirementSlice: slice } : {}),
|
|
3348
|
-
prompt: slice.length > 0
|
|
3349
|
-
? buildUxPrompt(slice)
|
|
3350
|
-
: buildPhasePrompt(uxSegment),
|
|
3351
|
-
}));
|
|
3352
|
-
for (const segment of FRONTEND_PLAN_SEGMENTS) {
|
|
3353
|
-
if (["coverage", "ux-local", "finalize"].includes(segment.id))
|
|
3354
|
-
continue;
|
|
4401
|
+
if (useCompactSmallPlan) {
|
|
4402
|
+
// Compact mode already queued both sessions above.
|
|
4403
|
+
}
|
|
4404
|
+
else if (!input.parallelCoverageOnly) {
|
|
4405
|
+
if (requirementIdsProvided) {
|
|
4406
|
+
queue.push({
|
|
4407
|
+
id: uxRegistrySegment.id,
|
|
4408
|
+
toolNames: uxRegistrySegment.toolNames,
|
|
4409
|
+
prompt: buildUxRegistryPrompt(),
|
|
4410
|
+
});
|
|
4411
|
+
}
|
|
4412
|
+
const uxSlice = requirementIdsProvided ? [...allRequirementIds] : [];
|
|
3355
4413
|
queue.push({
|
|
3356
|
-
id:
|
|
3357
|
-
|
|
3358
|
-
|
|
4414
|
+
id: "ux-local-1",
|
|
4415
|
+
// Compatibility path for an unreadable requirement inventory: there is
|
|
4416
|
+
// no safe all-requirements registry phase, so the unscoped UX session
|
|
4417
|
+
// must establish its registry before recording state flow.
|
|
4418
|
+
toolNames: requirementIdsProvided
|
|
4419
|
+
? uxSegment.toolNames
|
|
4420
|
+
: new Set([...(uxSegment.toolNames ?? []), "record_state_registry"]),
|
|
4421
|
+
...(uxSlice.length > 0 ? { requirementSlice: uxSlice } : {}),
|
|
4422
|
+
prompt: uxSlice.length > 0
|
|
4423
|
+
? buildUxPrompt(uxSlice)
|
|
4424
|
+
: buildPhasePrompt(uxSegment),
|
|
4425
|
+
});
|
|
4426
|
+
for (const segment of FRONTEND_PLAN_SEGMENTS) {
|
|
4427
|
+
if (["coverage", "ux-registry", "ux-local", "finalize"].includes(segment.id))
|
|
4428
|
+
continue;
|
|
4429
|
+
queue.push({
|
|
4430
|
+
id: segment.id,
|
|
4431
|
+
toolNames: segment.toolNames,
|
|
4432
|
+
prompt: buildPhasePrompt(segment),
|
|
4433
|
+
});
|
|
4434
|
+
}
|
|
4435
|
+
queue.push({
|
|
4436
|
+
id: finalizeSegment.id,
|
|
4437
|
+
toolNames: finalizeSegment.toolNames,
|
|
4438
|
+
prompt: buildPhasePrompt(finalizeSegment),
|
|
3359
4439
|
});
|
|
3360
4440
|
}
|
|
3361
|
-
|
|
3362
|
-
queue.push({
|
|
3363
|
-
id: finalizeSegment.id,
|
|
3364
|
-
toolNames: finalizeSegment.toolNames,
|
|
3365
|
-
prompt: buildPhasePrompt(finalizeSegment),
|
|
3366
|
-
});
|
|
3367
|
-
let last;
|
|
4441
|
+
let accumulated;
|
|
3368
4442
|
let index = 0;
|
|
3369
4443
|
let invocationCount = 0;
|
|
3370
4444
|
while (index < queue.length) {
|
|
@@ -3389,6 +4463,12 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3389
4463
|
const promptSlice = remaining.length > 0 ? remaining : session.coverageSlice;
|
|
3390
4464
|
prompt = buildCoveragePrompt(promptSlice, session.missingFacts ?? preexistingMissing);
|
|
3391
4465
|
}
|
|
4466
|
+
else if (session.id === "compact-local") {
|
|
4467
|
+
prompt = buildCompactLocalPrompt(session.missingFacts);
|
|
4468
|
+
}
|
|
4469
|
+
else if (session.id === "ux-registry") {
|
|
4470
|
+
prompt = buildUxRegistryPrompt(session.missingFacts);
|
|
4471
|
+
}
|
|
3392
4472
|
else if (session.id.startsWith("ux-local-")) {
|
|
3393
4473
|
// Build this at execution time: coverage facts are committed by the
|
|
3394
4474
|
// preceding sessions and must be visible to the UX-local model.
|
|
@@ -3402,12 +4482,17 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3402
4482
|
// stale queue entry from dropping facts written by earlier phases.
|
|
3403
4483
|
const segment = FRONTEND_PLAN_SEGMENTS.find((candidate) => candidate.id === session.id);
|
|
3404
4484
|
if (segment) {
|
|
3405
|
-
prompt =
|
|
4485
|
+
prompt =
|
|
4486
|
+
session.id === "finalize" && useCompactSmallPlan
|
|
4487
|
+
? buildCompactFinalizePrompt(session.missingFacts)
|
|
4488
|
+
: buildPhasePrompt(segment, session.missingFacts);
|
|
3406
4489
|
}
|
|
3407
4490
|
}
|
|
3408
4491
|
const committedBefore = input.committedFactCount();
|
|
3409
|
-
input.setActiveRequirementScope?.(session.
|
|
3410
|
-
|
|
4492
|
+
input.setActiveRequirementScope?.(session.coverageOnly ||
|
|
4493
|
+
session.id === "compact-local" ||
|
|
4494
|
+
session.id.startsWith("ux-local-")
|
|
4495
|
+
? session.coverageSlice ?? session.requirementSlice ?? []
|
|
3411
4496
|
: []);
|
|
3412
4497
|
if (invocationCount >= FRONTEND_PLAN_BATCH_MAX_SESSIONS)
|
|
3413
4498
|
break;
|
|
@@ -3425,7 +4510,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3425
4510
|
}
|
|
3426
4511
|
: {}),
|
|
3427
4512
|
});
|
|
3428
|
-
|
|
4513
|
+
accumulated = accumulated
|
|
4514
|
+
? combineSequentialPiResults(accumulated, result)
|
|
4515
|
+
: result;
|
|
3429
4516
|
try {
|
|
3430
4517
|
await input.flushLedger();
|
|
3431
4518
|
}
|
|
@@ -3444,27 +4531,81 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3444
4531
|
committedFacts: input.committedFacts(),
|
|
3445
4532
|
})
|
|
3446
4533
|
: [];
|
|
4534
|
+
const isCompactLocalSession = session.id === "compact-local";
|
|
4535
|
+
const isUxRegistrySession = session.id === "ux-registry";
|
|
3447
4536
|
const isUxLocalSession = session.id.startsWith("ux-local-");
|
|
3448
|
-
const missingPhase =
|
|
3449
|
-
?
|
|
3450
|
-
|
|
3451
|
-
|
|
3452
|
-
|
|
3453
|
-
|
|
3454
|
-
|
|
3455
|
-
|
|
4537
|
+
const missingPhase = isCompactLocalSession && input.committedFacts
|
|
4538
|
+
? [
|
|
4539
|
+
...collectFrontendPlanMissingFacts({
|
|
4540
|
+
requirementIds: allRequirementIds,
|
|
4541
|
+
committedFacts: input.committedFacts(),
|
|
4542
|
+
}),
|
|
4543
|
+
...collectFrontendPlanPhaseMissingFacts({
|
|
4544
|
+
phase: "ux-registry",
|
|
4545
|
+
requirementIds: allRequirementIds,
|
|
4546
|
+
committedFacts: input.committedFacts(),
|
|
4547
|
+
}),
|
|
4548
|
+
...collectFrontendPlanPhaseMissingFacts({
|
|
4549
|
+
phase: "ux-local",
|
|
4550
|
+
requirementIds: allRequirementIds,
|
|
4551
|
+
committedFacts: input.committedFacts(),
|
|
4552
|
+
behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
|
|
4553
|
+
}),
|
|
4554
|
+
]
|
|
4555
|
+
: isUxRegistrySession && input.committedFacts
|
|
3456
4556
|
? collectFrontendPlanPhaseMissingFacts({
|
|
3457
|
-
phase: "
|
|
4557
|
+
phase: "ux-registry",
|
|
3458
4558
|
requirementIds: allRequirementIds,
|
|
3459
4559
|
committedFacts: input.committedFacts(),
|
|
3460
4560
|
})
|
|
3461
|
-
:
|
|
4561
|
+
: isUxLocalSession && session.requirementSlice && input.committedFacts
|
|
4562
|
+
? collectFrontendPlanPhaseMissingFacts({
|
|
4563
|
+
phase: "ux-local",
|
|
4564
|
+
requirementIds: session.requirementSlice,
|
|
4565
|
+
committedFacts: input.committedFacts(),
|
|
4566
|
+
behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
|
|
4567
|
+
})
|
|
4568
|
+
: session.id === "global-mock-data" && allRequirementIds.length > 0 && input.committedFacts
|
|
4569
|
+
? collectFrontendPlanPhaseMissingFacts({
|
|
4570
|
+
phase: "global-mock-data",
|
|
4571
|
+
requirementIds: allRequirementIds,
|
|
4572
|
+
committedFacts: input.committedFacts(),
|
|
4573
|
+
})
|
|
4574
|
+
: [];
|
|
3462
4575
|
const missingPhaseFacts = [...missingCoverage, ...missingPhase];
|
|
4576
|
+
// Frozen verification commands must operate on files the planner has
|
|
4577
|
+
// committed verification targets for; otherwise the admission writeSet
|
|
4578
|
+
// cannot authorize the file and verify-shell deterministically fails.
|
|
4579
|
+
const verificationCommandFiles = collectFrontendVerificationCommandFiles(input.basePrompt);
|
|
4580
|
+
if (verificationCommandFiles.length > 0 && allRequirementIds.length > 0 && input.committedFacts) {
|
|
4581
|
+
const coveredVerificationFiles = new Set(input
|
|
4582
|
+
.committedFacts()
|
|
4583
|
+
.map(committedFactFromPlanRecord)
|
|
4584
|
+
.filter((fact) => Boolean(fact && fact.origin === "plan" && fact.kind === "plan-verification-target"))
|
|
4585
|
+
.map((fact) => fact.entry?.file)
|
|
4586
|
+
.filter((file) => typeof file === "string"));
|
|
4587
|
+
for (const file of verificationCommandFiles) {
|
|
4588
|
+
if (coveredVerificationFiles.has(file))
|
|
4589
|
+
continue;
|
|
4590
|
+
missingPhaseFacts.push({
|
|
4591
|
+
kind: "plan-verification-target",
|
|
4592
|
+
id: file,
|
|
4593
|
+
requirementIds: allRequirementIds.slice(0, 1),
|
|
4594
|
+
reason: `frozen verification command references ${file} but no committed verification target covers it; record a static verification target for this file so it joins the writeSet`,
|
|
4595
|
+
});
|
|
4596
|
+
}
|
|
4597
|
+
}
|
|
3463
4598
|
const promptForMissingPhase = (missing) => session.coverageOnly
|
|
3464
4599
|
? buildCoveragePrompt(session.coverageSlice ?? [], missing)
|
|
3465
|
-
:
|
|
3466
|
-
?
|
|
3467
|
-
:
|
|
4600
|
+
: isCompactLocalSession
|
|
4601
|
+
? buildCompactLocalPrompt(missing)
|
|
4602
|
+
: isUxRegistrySession
|
|
4603
|
+
? buildUxRegistryPrompt(missing)
|
|
4604
|
+
: session.id === "finalize" && useCompactSmallPlan
|
|
4605
|
+
? buildCompactFinalizePrompt(missing)
|
|
4606
|
+
: isUxLocalSession
|
|
4607
|
+
? buildUxPrompt(session.requirementSlice ?? [], missing)
|
|
4608
|
+
: buildPhasePrompt(globalMockDataSegment, missing);
|
|
3468
4609
|
if (result.ok) {
|
|
3469
4610
|
if (missingPhaseFacts.length === 0) {
|
|
3470
4611
|
index += 1;
|
|
@@ -3480,9 +4621,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3480
4621
|
continue;
|
|
3481
4622
|
}
|
|
3482
4623
|
return {
|
|
3483
|
-
...
|
|
4624
|
+
...accumulated,
|
|
3484
4625
|
ok: false,
|
|
3485
|
-
stderr: `${
|
|
4626
|
+
stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
|
|
3486
4627
|
failureCategory: "invalid-output",
|
|
3487
4628
|
};
|
|
3488
4629
|
}
|
|
@@ -3499,19 +4640,45 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3499
4640
|
}
|
|
3500
4641
|
if (missingPhaseFacts.length > 0) {
|
|
3501
4642
|
return {
|
|
3502
|
-
...
|
|
4643
|
+
...accumulated,
|
|
3503
4644
|
ok: false,
|
|
3504
|
-
stderr:
|
|
4645
|
+
stderr: `${accumulated.stderr}\nfrontend plan completeness check failed: ${missingPhaseFacts.map((item) => item.reason).join("; ")}`.trim(),
|
|
3505
4646
|
failureCategory: "invalid-output",
|
|
3506
4647
|
};
|
|
3507
4648
|
}
|
|
3508
4649
|
index += 1;
|
|
3509
4650
|
continue;
|
|
3510
4651
|
}
|
|
4652
|
+
// A length-stopped, fact-less compact session is the planner variant of
|
|
4653
|
+
// writer-thinking-exhausted. Give the same scope exactly one tool-first
|
|
4654
|
+
// retry before the split below re-batches the requirements, because one
|
|
4655
|
+
// reinforced full-scope pass is cheaper than re-planning split halves.
|
|
4656
|
+
const plannerThinkingBurn = isCompactLocalSession &&
|
|
4657
|
+
committedAfter === committedBefore &&
|
|
4658
|
+
!(result.assistantText ?? "").trim() &&
|
|
4659
|
+
!result.stderr.trim() &&
|
|
4660
|
+
!result.timedOut &&
|
|
4661
|
+
readWriterThinkingExhaustionEvidence(result).stopReason === "length";
|
|
4662
|
+
if (plannerThinkingBurn && (session.retryCount ?? 0) < 1) {
|
|
4663
|
+
queue[index] = {
|
|
4664
|
+
...session,
|
|
4665
|
+
retryCount: (session.retryCount ?? 0) + 1,
|
|
4666
|
+
prompt: buildCompactLocalPrompt(),
|
|
4667
|
+
};
|
|
4668
|
+
continue;
|
|
4669
|
+
}
|
|
3511
4670
|
// Option 5: a multi-requirement coverage batch that failed with ZERO
|
|
3512
4671
|
// new facts and no provider stderr is the upfront-reasoning burn —
|
|
3513
|
-
// halve the slice and retry instead of failing the attempt.
|
|
3514
|
-
|
|
4672
|
+
// halve the slice and retry instead of failing the attempt. The compact
|
|
4673
|
+
// local session participates through its requirementSlice so a
|
|
4674
|
+
// thinking-burned small plan degrades into smaller batched sessions
|
|
4675
|
+
// instead of replaying one full-scope prompt until the repair budget
|
|
4676
|
+
// runs out.
|
|
4677
|
+
const coverageSlice = session.coverageOnly
|
|
4678
|
+
? session.coverageSlice
|
|
4679
|
+
: isCompactLocalSession
|
|
4680
|
+
? session.requirementSlice
|
|
4681
|
+
: undefined;
|
|
3515
4682
|
const zeroProgressBurn = coverageSlice !== undefined &&
|
|
3516
4683
|
coverageSlice.length > 1 &&
|
|
3517
4684
|
committedAfter === committedBefore &&
|
|
@@ -3522,6 +4689,26 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3522
4689
|
const half = Math.ceil(coverageSlice.length / 2);
|
|
3523
4690
|
const firstSlice = coverageSlice.slice(0, half);
|
|
3524
4691
|
const secondSlice = coverageSlice.slice(half);
|
|
4692
|
+
if (isCompactLocalSession) {
|
|
4693
|
+
// The split halves leave compact mode: continue them as ordinary
|
|
4694
|
+
// coverage sessions so every downstream ladder branch applies.
|
|
4695
|
+
queue.splice(index, 1, {
|
|
4696
|
+
...session,
|
|
4697
|
+
id: `coverage-compact-split-1`,
|
|
4698
|
+
coverageOnly: true,
|
|
4699
|
+
coverageSlice: firstSlice,
|
|
4700
|
+
requirementSlice: firstSlice,
|
|
4701
|
+
prompt: buildCoveragePrompt(firstSlice),
|
|
4702
|
+
}, {
|
|
4703
|
+
...session,
|
|
4704
|
+
id: `coverage-compact-split-2`,
|
|
4705
|
+
coverageOnly: true,
|
|
4706
|
+
coverageSlice: secondSlice,
|
|
4707
|
+
requirementSlice: secondSlice,
|
|
4708
|
+
prompt: buildCoveragePrompt(secondSlice),
|
|
4709
|
+
});
|
|
4710
|
+
continue;
|
|
4711
|
+
}
|
|
3525
4712
|
queue.splice(index, 1, {
|
|
3526
4713
|
...session,
|
|
3527
4714
|
coverageSlice: firstSlice,
|
|
@@ -3533,7 +4720,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3533
4720
|
});
|
|
3534
4721
|
continue;
|
|
3535
4722
|
}
|
|
3536
|
-
if ((session.coverageOnly || isUxLocalSession || session.id === "global-mock-data") &&
|
|
4723
|
+
if ((session.coverageOnly || isUxRegistrySession || isUxLocalSession || session.id === "global-mock-data") &&
|
|
3537
4724
|
missingPhaseFacts.length > 0 &&
|
|
3538
4725
|
!result.stderr.trim() &&
|
|
3539
4726
|
!result.timedOut &&
|
|
@@ -3546,11 +4733,11 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3546
4733
|
};
|
|
3547
4734
|
continue;
|
|
3548
4735
|
}
|
|
3549
|
-
return
|
|
4736
|
+
return mapPlannerExhaustion(accumulated, committedAfter > committedBefore);
|
|
3550
4737
|
}
|
|
3551
4738
|
if (index < queue.length) {
|
|
3552
4739
|
return {
|
|
3553
|
-
...(
|
|
4740
|
+
...(accumulated ?? {
|
|
3554
4741
|
ok: false,
|
|
3555
4742
|
stdout: "",
|
|
3556
4743
|
stderr: "",
|
|
@@ -3567,17 +4754,195 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3567
4754
|
tokensUsed: 0,
|
|
3568
4755
|
}),
|
|
3569
4756
|
ok: false,
|
|
3570
|
-
stderr: `${
|
|
4757
|
+
stderr: `${accumulated?.stderr ?? ""}\nfrontend plan segmentation exceeded the ${FRONTEND_PLAN_BATCH_MAX_SESSIONS}-session safety limit before finalize`.trim(),
|
|
3571
4758
|
failureCategory: "invalid-output",
|
|
3572
4759
|
};
|
|
3573
4760
|
}
|
|
3574
|
-
return (
|
|
4761
|
+
return mapPlannerExhaustion(accumulated ?? {
|
|
3575
4762
|
ok: false,
|
|
3576
4763
|
stdout: "",
|
|
3577
4764
|
stderr: "frontend plan segmentation produced no session",
|
|
3578
4765
|
failureCategory: "empty-output",
|
|
3579
4766
|
durationMs: 0,
|
|
3580
|
-
});
|
|
4767
|
+
}, false);
|
|
4768
|
+
}
|
|
4769
|
+
/**
|
|
4770
|
+
* Run the two independent Scout evidence surfaces concurrently while keeping
|
|
4771
|
+
* their typed-event stores isolated. The main Scout store is the only store
|
|
4772
|
+
* visible to Plan; shard facts are merged in completion-order-independent
|
|
4773
|
+
* order after both sessions settle. This gives discovery real parallelism
|
|
4774
|
+
* without allowing sibling models to race a shared revision counter.
|
|
4775
|
+
*/
|
|
4776
|
+
function aggregateParallelPiResults(results) {
|
|
4777
|
+
const first = results[0];
|
|
4778
|
+
const failed = results.find((result) => !result.ok);
|
|
4779
|
+
const representative = failed ?? first;
|
|
4780
|
+
return {
|
|
4781
|
+
...representative,
|
|
4782
|
+
ok: failed === undefined,
|
|
4783
|
+
failureCategory: failed?.failureCategory ?? "success",
|
|
4784
|
+
durationMs: Math.max(...results.map((result) => result.durationMs), 0),
|
|
4785
|
+
exitCode: failed ? failed.exitCode : 0,
|
|
4786
|
+
stderr: results.map((result) => result.stderr).filter(Boolean).join("\n"),
|
|
4787
|
+
tokensUsed: results.reduce((total, result) => total + result.tokensUsed, 0),
|
|
4788
|
+
parsedEvents: results.reduce((total, result) => total + result.parsedEvents, 0),
|
|
4789
|
+
attemptedModels: [
|
|
4790
|
+
...new Set(results.flatMap((result) => result.attemptedModels)),
|
|
4791
|
+
],
|
|
4792
|
+
fallbackUsed: results.some((result) => result.fallbackUsed),
|
|
4793
|
+
timedOut: results.some((result) => result.timedOut),
|
|
4794
|
+
};
|
|
4795
|
+
}
|
|
4796
|
+
function combineSequentialPiResults(first, second) {
|
|
4797
|
+
return {
|
|
4798
|
+
...second,
|
|
4799
|
+
durationMs: first.durationMs + second.durationMs,
|
|
4800
|
+
stderr: [first.stderr, second.stderr].filter(Boolean).join("\n"),
|
|
4801
|
+
tokensUsed: first.tokensUsed + second.tokensUsed,
|
|
4802
|
+
parsedEvents: first.parsedEvents + second.parsedEvents,
|
|
4803
|
+
attemptedModels: [
|
|
4804
|
+
...new Set([...first.attemptedModels, ...second.attemptedModels]),
|
|
4805
|
+
],
|
|
4806
|
+
fallbackUsed: first.fallbackUsed || second.fallbackUsed,
|
|
4807
|
+
timedOut: first.timedOut || second.timedOut,
|
|
4808
|
+
};
|
|
4809
|
+
}
|
|
4810
|
+
async function runFrontendScoutParallelSessions(input) {
|
|
4811
|
+
const [{ createTypedEventStore }] = await Promise.all([
|
|
4812
|
+
import("../workflows/dag/frontend-typed-event-store.js"),
|
|
4813
|
+
]);
|
|
4814
|
+
const shards = [
|
|
4815
|
+
{
|
|
4816
|
+
id: "surface",
|
|
4817
|
+
toolName: "record_target_surface",
|
|
4818
|
+
instruction: [
|
|
4819
|
+
"PARALLEL SCOUT SHARD — target surface only.",
|
|
4820
|
+
"Inspect routes, entrypoints, implementation ownership, data source, and applicable test paths.",
|
|
4821
|
+
"Call record_target_surface exactly once with the complete runtime-evidenced surface. Do not call record_design_evidence.",
|
|
4822
|
+
].join(" "),
|
|
4823
|
+
},
|
|
4824
|
+
{
|
|
4825
|
+
id: "design",
|
|
4826
|
+
toolName: "record_design_evidence",
|
|
4827
|
+
instruction: [
|
|
4828
|
+
"PARALLEL SCOUT SHARD — design evidence only.",
|
|
4829
|
+
"Inspect the frontend framework, styling/theme conventions, reusable components, and relevant design/spec files.",
|
|
4830
|
+
"Call record_design_evidence for the evidence you actually read. Do not call record_target_surface.",
|
|
4831
|
+
].join(" "),
|
|
4832
|
+
},
|
|
4833
|
+
];
|
|
4834
|
+
const outcomes = await Promise.all(shards.map(async (shard) => {
|
|
4835
|
+
let shardTools;
|
|
4836
|
+
let shardResult;
|
|
4837
|
+
try {
|
|
4838
|
+
const store = createTypedEventStore();
|
|
4839
|
+
const shardNodeId = `${input.nodeId}/parallel/${shard.id}`;
|
|
4840
|
+
shardTools = await createFrontendScoutEvidenceTools({
|
|
4841
|
+
attemptId: `${input.attemptId}:parallel:${shard.id}`,
|
|
4842
|
+
store,
|
|
4843
|
+
runDir: input.runDir,
|
|
4844
|
+
nodeId: shardNodeId,
|
|
4845
|
+
workspaceRoot: input.workspaceRoot,
|
|
4846
|
+
sourceDeclaredPaths: input.sourceDeclaredPaths,
|
|
4847
|
+
});
|
|
4848
|
+
const customTools = shardTools.customTools.filter((tool) => typeof tool === "object" &&
|
|
4849
|
+
tool !== null &&
|
|
4850
|
+
tool.name === shard.toolName);
|
|
4851
|
+
const sessionOptions = {
|
|
4852
|
+
...input.sessionOptions,
|
|
4853
|
+
sessionEventsPath: path.join(input.runDir, shardNodeId, "session-events.jsonl"),
|
|
4854
|
+
};
|
|
4855
|
+
shardResult = await input.piStepFn({
|
|
4856
|
+
...sessionOptions,
|
|
4857
|
+
prompt: `${input.basePrompt}\n\n${shard.instruction}`,
|
|
4858
|
+
writerToolPolicy: {
|
|
4859
|
+
requireSdk: true,
|
|
4860
|
+
customTools: [...customTools, ...(input.readBudgetTools ?? [])],
|
|
4861
|
+
},
|
|
4862
|
+
});
|
|
4863
|
+
await shardTools.flush();
|
|
4864
|
+
return { shard, result: shardResult, tools: shardTools };
|
|
4865
|
+
}
|
|
4866
|
+
catch (error) {
|
|
4867
|
+
const crashMessage = `frontend scout parallel shard ${shard.id} crashed: ${error instanceof Error ? error.message : String(error)}`;
|
|
4868
|
+
return {
|
|
4869
|
+
shard,
|
|
4870
|
+
tools: shardTools,
|
|
4871
|
+
result: shardResult
|
|
4872
|
+
? {
|
|
4873
|
+
...shardResult,
|
|
4874
|
+
ok: false,
|
|
4875
|
+
failureCategory: shardResult.ok
|
|
4876
|
+
? "invalid-output"
|
|
4877
|
+
: shardResult.failureCategory,
|
|
4878
|
+
stderr: [shardResult.stderr, crashMessage]
|
|
4879
|
+
.filter(Boolean)
|
|
4880
|
+
.join("\n"),
|
|
4881
|
+
}
|
|
4882
|
+
: {
|
|
4883
|
+
ok: false,
|
|
4884
|
+
assistantText: "",
|
|
4885
|
+
command: [],
|
|
4886
|
+
durationMs: 0,
|
|
4887
|
+
exitCode: null,
|
|
4888
|
+
failureCategory: "tool-policy",
|
|
4889
|
+
modelDisplay: "unknown",
|
|
4890
|
+
parsedEvents: 0,
|
|
4891
|
+
stderr: crashMessage,
|
|
4892
|
+
stdout: "",
|
|
4893
|
+
timedOut: false,
|
|
4894
|
+
attemptedModels: [],
|
|
4895
|
+
fallbackUsed: false,
|
|
4896
|
+
tokensUsed: 0,
|
|
4897
|
+
},
|
|
4898
|
+
};
|
|
4899
|
+
}
|
|
4900
|
+
}));
|
|
4901
|
+
const ordered = [...outcomes].sort((left, right) => left.shard.id.localeCompare(right.shard.id));
|
|
4902
|
+
try {
|
|
4903
|
+
for (const outcome of ordered) {
|
|
4904
|
+
if (outcome.result.ok && outcome.tools) {
|
|
4905
|
+
await input.mainTools.adoptCommittedFacts(outcome.tools.committedFacts());
|
|
4906
|
+
}
|
|
4907
|
+
}
|
|
4908
|
+
}
|
|
4909
|
+
catch (error) {
|
|
4910
|
+
const aggregate = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
|
|
4911
|
+
return {
|
|
4912
|
+
...aggregate,
|
|
4913
|
+
ok: false,
|
|
4914
|
+
failureCategory: "invalid-output",
|
|
4915
|
+
stderr: [
|
|
4916
|
+
aggregate.stderr,
|
|
4917
|
+
`frontend scout parallel fact merge failed: ${error instanceof Error ? error.message : String(error)}`,
|
|
4918
|
+
]
|
|
4919
|
+
.filter(Boolean)
|
|
4920
|
+
.join("\n"),
|
|
4921
|
+
};
|
|
4922
|
+
}
|
|
4923
|
+
const failed = ordered.find((outcome) => !outcome.result.ok);
|
|
4924
|
+
if (failed) {
|
|
4925
|
+
const aggregate = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
|
|
4926
|
+
return {
|
|
4927
|
+
...aggregate,
|
|
4928
|
+
ok: false,
|
|
4929
|
+
stderr: `${aggregate.stderr}\nfrontend scout parallel shard failed: ${failed.shard.id}`.trim(),
|
|
4930
|
+
};
|
|
4931
|
+
}
|
|
4932
|
+
const committedKinds = new Set(ordered.flatMap((outcome) => (outcome.tools?.committedFacts() ?? [])
|
|
4933
|
+
.map((record) => record.fact?.kind)
|
|
4934
|
+
.filter((kind) => typeof kind === "string")));
|
|
4935
|
+
const missing = ["target-surface", "design-evidence"].filter((kind) => !committedKinds.has(kind));
|
|
4936
|
+
if (missing.length > 0) {
|
|
4937
|
+
const fallback = aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
|
|
4938
|
+
return {
|
|
4939
|
+
...fallback,
|
|
4940
|
+
ok: false,
|
|
4941
|
+
failureCategory: "invalid-output",
|
|
4942
|
+
stderr: `${fallback.stderr}\nfrontend scout parallel shards committed no ${missing.join(" or ")} fact`.trim(),
|
|
4943
|
+
};
|
|
4944
|
+
}
|
|
4945
|
+
return aggregateParallelPiResults(ordered.map((outcome) => outcome.result));
|
|
3581
4946
|
}
|
|
3582
4947
|
export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
|
|
3583
4948
|
const started = Date.now();
|
|
@@ -3673,6 +5038,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
3673
5038
|
let planLedgerTools;
|
|
3674
5039
|
let contractTools;
|
|
3675
5040
|
let scoutEvidenceTools;
|
|
5041
|
+
let scoutSourceDeclaredPaths;
|
|
3676
5042
|
let readBudgetTools;
|
|
3677
5043
|
const commandPolicy = resolveDagCommandPolicy(input.task.commandPolicy);
|
|
3678
5044
|
const allowsPlaywrightCli = dagCommandPolicyAllows(input.task.commandPolicy, "playwright-cli");
|
|
@@ -3816,6 +5182,10 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
3816
5182
|
store,
|
|
3817
5183
|
runDir: meta.runDir,
|
|
3818
5184
|
nodeId: input.task.id,
|
|
5185
|
+
canonicalRequirements: await resolveFrontendCanonicalRequirements({
|
|
5186
|
+
cwd: input.cwd,
|
|
5187
|
+
sourceBinding: meta.spec.sourceBinding,
|
|
5188
|
+
}),
|
|
3819
5189
|
});
|
|
3820
5190
|
writerToolPolicy = {
|
|
3821
5191
|
requireSdk: true,
|
|
@@ -3836,16 +5206,17 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
3836
5206
|
try {
|
|
3837
5207
|
const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
|
|
3838
5208
|
const store = createTypedEventStore();
|
|
5209
|
+
scoutSourceDeclaredPaths = await resolveFrontendScoutSourceDeclaredPaths({
|
|
5210
|
+
cwd: input.cwd,
|
|
5211
|
+
spec: meta.spec,
|
|
5212
|
+
});
|
|
3839
5213
|
scoutEvidenceTools = await createFrontendScoutEvidenceTools({
|
|
3840
5214
|
attemptId: `${meta.runId}:${input.task.id}`,
|
|
3841
5215
|
store,
|
|
3842
5216
|
runDir: meta.runDir,
|
|
3843
5217
|
nodeId: input.task.id,
|
|
3844
5218
|
workspaceRoot: input.cwd,
|
|
3845
|
-
sourceDeclaredPaths:
|
|
3846
|
-
cwd: input.cwd,
|
|
3847
|
-
spec: meta.spec,
|
|
3848
|
-
}),
|
|
5219
|
+
sourceDeclaredPaths: scoutSourceDeclaredPaths,
|
|
3849
5220
|
});
|
|
3850
5221
|
writerToolPolicy = {
|
|
3851
5222
|
requireSdk: true,
|
|
@@ -3878,6 +5249,13 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
3878
5249
|
cwd: input.cwd,
|
|
3879
5250
|
sourceBinding: meta.spec.sourceBinding,
|
|
3880
5251
|
}),
|
|
5252
|
+
declaredUiStateIds: await resolveFrontendDeclaredUiStateIds({
|
|
5253
|
+
runDir: meta.runDir,
|
|
5254
|
+
}),
|
|
5255
|
+
canonicalVerificationTargetIds: await resolveFrontendCanonicalVerificationTargetIds({
|
|
5256
|
+
runDir: meta.runDir,
|
|
5257
|
+
}),
|
|
5258
|
+
workspaceRoot: input.cwd,
|
|
3881
5259
|
});
|
|
3882
5260
|
writerToolPolicy = {
|
|
3883
5261
|
requireSdk: true,
|
|
@@ -4017,11 +5395,27 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4017
5395
|
}
|
|
4018
5396
|
: undefined,
|
|
4019
5397
|
};
|
|
4020
|
-
if (
|
|
4021
|
-
|
|
4022
|
-
|
|
4023
|
-
|
|
4024
|
-
|
|
5398
|
+
if (isFrontendScoutEvidenceNode(input.task) &&
|
|
5399
|
+
scoutEvidenceTools &&
|
|
5400
|
+
input.task.complexity !== "LOW") {
|
|
5401
|
+
result = await runFrontendScoutParallelSessions({
|
|
5402
|
+
piStepFn,
|
|
5403
|
+
sessionOptions: piSessionOptions,
|
|
5404
|
+
basePrompt: input.prompt,
|
|
5405
|
+
mainTools: scoutEvidenceTools,
|
|
5406
|
+
runDir: meta.runDir,
|
|
5407
|
+
nodeId: input.task.id,
|
|
5408
|
+
attemptId: `${meta.runId}:${input.task.id}`,
|
|
5409
|
+
workspaceRoot: input.cwd,
|
|
5410
|
+
sourceDeclaredPaths: scoutSourceDeclaredPaths,
|
|
5411
|
+
readBudgetTools: readBudgetTools?.customTools,
|
|
5412
|
+
});
|
|
5413
|
+
}
|
|
5414
|
+
else if (isFrontendPlanLedgerNode(input.task) && planLedgerTools) {
|
|
5415
|
+
// Frontend-only split: independent coverage map sessions feed a single
|
|
5416
|
+
// reducer (UX decisions -> global policy -> finalize), mirroring the
|
|
5417
|
+
// backend-test template's module sharding. Small plans normally have one
|
|
5418
|
+
// coverage batch and retain the compact path below.
|
|
4025
5419
|
let planRequirementIds = [];
|
|
4026
5420
|
const planRequirementCosts = new Map();
|
|
4027
5421
|
const behaviorRequiredRequirementIds = [];
|
|
@@ -4039,12 +5433,8 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4039
5433
|
?.kind === "requirement")
|
|
4040
5434
|
.map((record) => record.fact?.id)
|
|
4041
5435
|
.filter((id) => typeof id === "string");
|
|
4042
|
-
for (const
|
|
4043
|
-
|
|
4044
|
-
if (fact?.kind !== "requirement" || typeof fact.id !== "string")
|
|
4045
|
-
continue;
|
|
4046
|
-
const evidence = fact.evidence;
|
|
4047
|
-
if (evidence?.behavior === "required")
|
|
5436
|
+
for (const fact of resolveFrontendContractRequirements(contractFacts.map((record) => record.fact))) {
|
|
5437
|
+
if (fact.evidence.behavior === "required")
|
|
4048
5438
|
behaviorRequiredRequirementIds.push(fact.id);
|
|
4049
5439
|
planRequirementCosts.set(fact.id, estimateFrontendPlanRequirementRecordCalls(fact));
|
|
4050
5440
|
}
|
|
@@ -4052,25 +5442,189 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4052
5442
|
catch {
|
|
4053
5443
|
// Unreadable ledger falls back to a single coverage session.
|
|
4054
5444
|
}
|
|
4055
|
-
|
|
5445
|
+
// Independent requirement-coverage batches are map workers. Each worker
|
|
5446
|
+
// owns an isolated typed-event store; only after all workers settle do we
|
|
5447
|
+
// merge facts into the main Plan ledger and run the single UX/global/
|
|
5448
|
+
// finalize reducer. This avoids revision races while shortening the
|
|
5449
|
+
// longest coverage phase for large plans.
|
|
5450
|
+
const runPlanSessions = (options) => runFrontendPlanSegmentedSessions({
|
|
4056
5451
|
piStepFn,
|
|
4057
|
-
sessionOptions:
|
|
4058
|
-
basePrompt: input.prompt,
|
|
5452
|
+
sessionOptions: options.sessionOptions,
|
|
5453
|
+
basePrompt: options.basePrompt ?? input.prompt,
|
|
4059
5454
|
attempt: input.attempt ?? 1,
|
|
4060
|
-
committedFactCount: () =>
|
|
4061
|
-
requirementIds: planRequirementIds,
|
|
5455
|
+
committedFactCount: () => options.ledgerTools.committedFactCount(),
|
|
5456
|
+
requirementIds: options.requirementIds ?? planRequirementIds,
|
|
4062
5457
|
requirementCosts: planRequirementCosts,
|
|
4063
|
-
|
|
4064
|
-
|
|
5458
|
+
...(options.parallelCoverageOnly !== undefined
|
|
5459
|
+
? { parallelCoverageOnly: options.parallelCoverageOnly }
|
|
5460
|
+
: {}),
|
|
5461
|
+
...(options.compactSmallPlan !== undefined
|
|
5462
|
+
? { compactSmallPlan: options.compactSmallPlan }
|
|
5463
|
+
: {}),
|
|
5464
|
+
committedRequirementIds: () => options.ledgerTools.committedRequirementIds(),
|
|
5465
|
+
committedFacts: () => options.ledgerTools.committedFacts(),
|
|
4065
5466
|
behaviorRequiredRequirementIds,
|
|
4066
|
-
setActiveRequirementScope: (requirementIds) =>
|
|
5467
|
+
setActiveRequirementScope: (requirementIds) => options.ledgerTools.setActiveRequirementScope(requirementIds),
|
|
4067
5468
|
segmentCustomTools: (toolNames) => toolNames === null
|
|
4068
|
-
?
|
|
4069
|
-
:
|
|
5469
|
+
? options.ledgerTools.customTools
|
|
5470
|
+
: options.ledgerTools.customTools.filter((tool) => typeof tool === "object" &&
|
|
4070
5471
|
tool !== null &&
|
|
4071
5472
|
toolNames.has(tool.name)),
|
|
4072
|
-
flushLedger: () =>
|
|
5473
|
+
flushLedger: () => options.ledgerTools.flush(),
|
|
4073
5474
|
});
|
|
5475
|
+
if (planRequirementIds.length > 1) {
|
|
5476
|
+
const estimatedPlanCalls = planRequirementIds.reduce((total, id) => total + Math.max(1, planRequirementCosts.get(id) ?? 2), 0);
|
|
5477
|
+
const compactEligibleBeforeSharding = planRequirementIds.length <= FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS &&
|
|
5478
|
+
estimatedPlanCalls > 12 &&
|
|
5479
|
+
estimatedPlanCalls <= FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS &&
|
|
5480
|
+
countFrontendPlanTargetSurfaces(input.prompt) === 1;
|
|
5481
|
+
if (compactEligibleBeforeSharding) {
|
|
5482
|
+
// Decide the small-plan topology before creating coverage shards.
|
|
5483
|
+
// Sharding first would make the compact two-session path unreachable.
|
|
5484
|
+
result = await runPlanSessions({
|
|
5485
|
+
ledgerTools: planLedgerTools,
|
|
5486
|
+
sessionOptions: piSessionOptions,
|
|
5487
|
+
compactSmallPlan: true,
|
|
5488
|
+
});
|
|
5489
|
+
}
|
|
5490
|
+
else {
|
|
5491
|
+
const coverageBatches = batchFrontendPlanRequirements({
|
|
5492
|
+
requirementIds: planRequirementIds,
|
|
5493
|
+
maxEstimatedRecordCalls: FRONTEND_PLAN_COVERAGE_MAX_RECORD_CALLS,
|
|
5494
|
+
maxRequirements: 4,
|
|
5495
|
+
requirementCosts: planRequirementCosts,
|
|
5496
|
+
});
|
|
5497
|
+
if (coverageBatches.length > 1) {
|
|
5498
|
+
const shardResults = await mapWithConcurrency(coverageBatches, FRONTEND_PLAN_COVERAGE_MAX_CONCURRENCY, async (slice, index) => {
|
|
5499
|
+
let shardTools;
|
|
5500
|
+
let shardResult;
|
|
5501
|
+
try {
|
|
5502
|
+
const { createTypedEventStore } = await import("../workflows/dag/frontend-typed-event-store.js");
|
|
5503
|
+
shardTools = await createFrontendPlanLedgerTools({
|
|
5504
|
+
attemptId: `${meta.runId}:${input.task.id}:parallel:${index + 1}`,
|
|
5505
|
+
store: createTypedEventStore(),
|
|
5506
|
+
runDir: meta.runDir,
|
|
5507
|
+
nodeId: `${input.task.id}/parallel/coverage-${index + 1}`,
|
|
5508
|
+
skeleton: input.task.structuredContractOutput?.skeleton,
|
|
5509
|
+
sourceBinding: meta.spec.sourceBinding,
|
|
5510
|
+
requirementIds: slice,
|
|
5511
|
+
writeSetPatterns: input.task.writeSet,
|
|
5512
|
+
canonicalVerificationTargetIds: await resolveFrontendCanonicalVerificationTargetIds({
|
|
5513
|
+
runDir: meta.runDir,
|
|
5514
|
+
}),
|
|
5515
|
+
componentNewSourceReferences: await resolveFrontendPlanNewComponentSourceReferences({
|
|
5516
|
+
cwd: input.cwd,
|
|
5517
|
+
sourceBinding: meta.spec.sourceBinding,
|
|
5518
|
+
}),
|
|
5519
|
+
});
|
|
5520
|
+
const shardSessionOptions = {
|
|
5521
|
+
...piSessionOptions,
|
|
5522
|
+
sessionEventsPath: path.join(meta.runDir, input.task.id, "parallel", `coverage-${index + 1}`, "session-events.jsonl"),
|
|
5523
|
+
};
|
|
5524
|
+
shardResult = await runPlanSessions({
|
|
5525
|
+
ledgerTools: shardTools,
|
|
5526
|
+
sessionOptions: shardSessionOptions,
|
|
5527
|
+
requirementIds: slice,
|
|
5528
|
+
basePrompt: `${input.prompt}\n\nPARALLEL COVERAGE SHARD ${index + 1}: use the frozen canonical behavior verification-target ids listed in the record_plan_verification_target tool description — do not prefix ids with a shard namespace or invent variant ids; identical cross-shard targets are deduped, divergent ones fail the merge.`,
|
|
5529
|
+
parallelCoverageOnly: true,
|
|
5530
|
+
});
|
|
5531
|
+
await shardTools.flush();
|
|
5532
|
+
return { index, result: shardResult, tools: shardTools };
|
|
5533
|
+
}
|
|
5534
|
+
catch (error) {
|
|
5535
|
+
const crashMessage = `frontend plan coverage shard ${index + 1} crashed: ${error instanceof Error ? error.message : String(error)}`;
|
|
5536
|
+
return {
|
|
5537
|
+
index,
|
|
5538
|
+
tools: shardTools,
|
|
5539
|
+
result: shardResult
|
|
5540
|
+
? {
|
|
5541
|
+
...shardResult,
|
|
5542
|
+
ok: false,
|
|
5543
|
+
failureCategory: shardResult.ok
|
|
5544
|
+
? "invalid-output"
|
|
5545
|
+
: shardResult.failureCategory,
|
|
5546
|
+
stderr: [shardResult.stderr, crashMessage]
|
|
5547
|
+
.filter(Boolean)
|
|
5548
|
+
.join("\n"),
|
|
5549
|
+
}
|
|
5550
|
+
: {
|
|
5551
|
+
ok: false,
|
|
5552
|
+
assistantText: "",
|
|
5553
|
+
command: [],
|
|
5554
|
+
durationMs: 0,
|
|
5555
|
+
exitCode: null,
|
|
5556
|
+
failureCategory: "tool-policy",
|
|
5557
|
+
modelDisplay: "unknown",
|
|
5558
|
+
parsedEvents: 0,
|
|
5559
|
+
stderr: crashMessage,
|
|
5560
|
+
stdout: "",
|
|
5561
|
+
timedOut: false,
|
|
5562
|
+
attemptedModels: [],
|
|
5563
|
+
fallbackUsed: false,
|
|
5564
|
+
tokensUsed: 0,
|
|
5565
|
+
},
|
|
5566
|
+
};
|
|
5567
|
+
}
|
|
5568
|
+
});
|
|
5569
|
+
const coverageResult = aggregateParallelPiResults(shardResults.map((shard) => shard.result));
|
|
5570
|
+
let mergeFailure;
|
|
5571
|
+
try {
|
|
5572
|
+
// Flatten every shard's facts FIRST, then pre-reduce: shards
|
|
5573
|
+
// partition requirements but a frozen verification target can
|
|
5574
|
+
// span shards, so same-id targets must merge across shards
|
|
5575
|
+
// before adoption, not per shard.
|
|
5576
|
+
await planLedgerTools.adoptCommittedFacts(reduceParallelCoverageShardRecords(shardResults
|
|
5577
|
+
.filter((item) => item.result.ok && item.tools)
|
|
5578
|
+
.flatMap((shard) => normalizeParallelCoverageShardRecords(shard.tools.committedFacts(), shard.index + 1))));
|
|
5579
|
+
}
|
|
5580
|
+
catch (error) {
|
|
5581
|
+
mergeFailure = error;
|
|
5582
|
+
}
|
|
5583
|
+
const failedShard = shardResults.find((shard) => !shard.result.ok);
|
|
5584
|
+
if (mergeFailure) {
|
|
5585
|
+
result = {
|
|
5586
|
+
...coverageResult,
|
|
5587
|
+
ok: false,
|
|
5588
|
+
failureCategory: "invalid-output",
|
|
5589
|
+
stderr: [
|
|
5590
|
+
coverageResult.stderr,
|
|
5591
|
+
`frontend plan coverage shard merge failed: ${mergeFailure instanceof Error ? mergeFailure.message : String(mergeFailure)}`,
|
|
5592
|
+
]
|
|
5593
|
+
.filter(Boolean)
|
|
5594
|
+
.join("\n"),
|
|
5595
|
+
};
|
|
5596
|
+
}
|
|
5597
|
+
else if (failedShard) {
|
|
5598
|
+
result = {
|
|
5599
|
+
...coverageResult,
|
|
5600
|
+
ok: false,
|
|
5601
|
+
stderr: `${coverageResult.stderr}\nfrontend plan coverage shard ${failedShard.index + 1} failed before reduce`.trim(),
|
|
5602
|
+
};
|
|
5603
|
+
}
|
|
5604
|
+
else {
|
|
5605
|
+
const reducerResult = await runPlanSessions({
|
|
5606
|
+
ledgerTools: planLedgerTools,
|
|
5607
|
+
sessionOptions: piSessionOptions,
|
|
5608
|
+
});
|
|
5609
|
+
result = combineSequentialPiResults(coverageResult, reducerResult);
|
|
5610
|
+
}
|
|
5611
|
+
}
|
|
5612
|
+
else {
|
|
5613
|
+
result = await runPlanSessions({
|
|
5614
|
+
ledgerTools: planLedgerTools,
|
|
5615
|
+
sessionOptions: piSessionOptions,
|
|
5616
|
+
compactSmallPlan: true,
|
|
5617
|
+
});
|
|
5618
|
+
}
|
|
5619
|
+
}
|
|
5620
|
+
}
|
|
5621
|
+
else {
|
|
5622
|
+
result = await runPlanSessions({
|
|
5623
|
+
ledgerTools: planLedgerTools,
|
|
5624
|
+
sessionOptions: piSessionOptions,
|
|
5625
|
+
compactSmallPlan: true,
|
|
5626
|
+
});
|
|
5627
|
+
}
|
|
4074
5628
|
}
|
|
4075
5629
|
else {
|
|
4076
5630
|
result = await piStepFn({
|
|
@@ -4079,6 +5633,18 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4079
5633
|
...(writerToolPolicy ? { writerToolPolicy } : {}),
|
|
4080
5634
|
});
|
|
4081
5635
|
}
|
|
5636
|
+
try {
|
|
5637
|
+
const verificationCommandFiles = collectFrontendVerificationCommandFiles(input.prompt);
|
|
5638
|
+
const debugPath = path.join(meta.runDir, input.task.id, "verification-command-coverage.json");
|
|
5639
|
+
await import("node:fs/promises").then(({ writeFile, mkdir }) => mkdir(path.dirname(debugPath), { recursive: true }).then(() => writeFile(debugPath, `${JSON.stringify({
|
|
5640
|
+
schemaVersion: 1,
|
|
5641
|
+
basePromptChars: input.prompt.length,
|
|
5642
|
+
verificationCommandFiles,
|
|
5643
|
+
}, null, 2)}\n`)));
|
|
5644
|
+
}
|
|
5645
|
+
catch {
|
|
5646
|
+
// best-effort diagnostic breadcrumb
|
|
5647
|
+
}
|
|
4082
5648
|
}
|
|
4083
5649
|
catch (error) {
|
|
4084
5650
|
if (playwrightToolContext) {
|
|
@@ -4121,6 +5687,37 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4121
5687
|
: input.task.writerOutcomePolicy
|
|
4122
5688
|
? WRITER_OUTCOME_PROTOCOL_LINE
|
|
4123
5689
|
: input.task.firstProtocolLine);
|
|
5690
|
+
if (input.task.id === "generate-backend-md-plan-pi") {
|
|
5691
|
+
if (mapped.stopReason === "length") {
|
|
5692
|
+
return {
|
|
5693
|
+
...mapped,
|
|
5694
|
+
ok: false,
|
|
5695
|
+
failureCategory: STRUCTURED_OUTPUT_RETRY_CATEGORY,
|
|
5696
|
+
stderr: [
|
|
5697
|
+
mapped.stderr,
|
|
5698
|
+
"backend-test Markdown plan was truncated before its required protocol could be trusted: stopReason=length; retry with the final artifact first",
|
|
5699
|
+
]
|
|
5700
|
+
.filter(Boolean)
|
|
5701
|
+
.join("\n"),
|
|
5702
|
+
};
|
|
5703
|
+
}
|
|
5704
|
+
if (mapped.ok) {
|
|
5705
|
+
const protocol = assessBackendTestPlanProtocol(mapped.assistantText || mapped.stdout);
|
|
5706
|
+
if (!protocol.ok) {
|
|
5707
|
+
return {
|
|
5708
|
+
...mapped,
|
|
5709
|
+
ok: false,
|
|
5710
|
+
failureCategory: "invalid-output",
|
|
5711
|
+
stderr: [
|
|
5712
|
+
mapped.stderr,
|
|
5713
|
+
`backend-test Markdown plan protocol invalid: ${protocol.issues.map((issue) => `${issue.code}: ${issue.detail}`).join("; ")}`,
|
|
5714
|
+
]
|
|
5715
|
+
.filter(Boolean)
|
|
5716
|
+
.join("\n"),
|
|
5717
|
+
};
|
|
5718
|
+
}
|
|
5719
|
+
}
|
|
5720
|
+
}
|
|
4124
5721
|
// A node-specific budget is enforced for every frontend reader that opts in,
|
|
4125
5722
|
// including design/final review. Earlier code only checked contract, scout,
|
|
4126
5723
|
// and plan nodes, leaving the two largest review sessions unbounded.
|
|
@@ -4574,7 +6171,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4574
6171
|
runDir: meta.runDir,
|
|
4575
6172
|
progress,
|
|
4576
6173
|
attempt: input.attempt ?? 1,
|
|
4577
|
-
maxAttempts: input.task.retryPolicy?.maxAttempts ??
|
|
6174
|
+
maxAttempts: input.task.retryPolicy?.maxAttempts ?? 5,
|
|
4578
6175
|
});
|
|
4579
6176
|
if (progress.status !== "PASS") {
|
|
4580
6177
|
const classified = classifyBackendTestWriterCompletenessFailure(progress);
|
|
@@ -4624,7 +6221,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4624
6221
|
stderrParts.push(completenessFailure.detail);
|
|
4625
6222
|
}
|
|
4626
6223
|
if (writerThinkingExhausted) {
|
|
4627
|
-
stderrParts.push(`${
|
|
6224
|
+
stderrParts.push(`${OUTPUT_LIMIT_RETRY_CATEGORY}: stopReason=length ended the turn before any write tool call; retry from the existing workspace and complete only unfinished targets`);
|
|
4628
6225
|
}
|
|
4629
6226
|
if (meta.writeGuardAttribution === "best-effort") {
|
|
4630
6227
|
stderrParts.push("write guard note: concurrent rank writers use best-effort per-node attribution; keep same-rank writeSet entries disjoint");
|
|
@@ -4697,14 +6294,11 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4697
6294
|
rawFailureCategory === "context-overflow" &&
|
|
4698
6295
|
(changeManifestChangedFiles?.length ?? 0) > 0
|
|
4699
6296
|
? "partial-success-with-context-overflow"
|
|
4700
|
-
: //
|
|
4701
|
-
//
|
|
4702
|
-
//
|
|
4703
|
-
// the empty-output base category so provider/transport failures keep
|
|
4704
|
-
// their original category, and only when the completeness gate did not
|
|
4705
|
-
// upgrade to incomplete-write-set above.
|
|
6297
|
+
: // output-limit: a length-stopped attempt is capacity truncation, not
|
|
6298
|
+
// empty output. Preserve the raw category for diagnostics and let the
|
|
6299
|
+
// node retry from committed facts/current workspace state.
|
|
4706
6300
|
writerThinkingExhausted
|
|
4707
|
-
?
|
|
6301
|
+
? OUTPUT_LIMIT_RETRY_CATEGORY
|
|
4708
6302
|
: writerCleanTimeout
|
|
4709
6303
|
? WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY
|
|
4710
6304
|
: // writer-budget-exhausted: the provider session consumed an
|
|
@@ -4860,6 +6454,8 @@ export function mapPiResultToDagNodeResult(result, firstProtocolLine) {
|
|
|
4860
6454
|
tokensUsed: result.tokensUsed,
|
|
4861
6455
|
parsedEvents: result.parsedEvents,
|
|
4862
6456
|
stopReason: readWriterThinkingExhaustionEvidence(result).stopReason,
|
|
6457
|
+
thinkingObserved: readWriterThinkingExhaustionEvidence(result).thinkingObserved,
|
|
6458
|
+
writeToolCallCount: readWriterThinkingExhaustionEvidence(result).writeToolCallCount,
|
|
4863
6459
|
};
|
|
4864
6460
|
}
|
|
4865
6461
|
function canonicalizeProtocolFirstLine(assistantText, firstProtocolLine) {
|