@tea-agent/loop-agent 0.43.0-next.18 → 0.43.0-next.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +1 -1
- package/CHANGELOG.md +60 -2308
- package/README.md +7 -0
- package/dist/application/task-lifecycle/advance.js +38 -7
- package/dist/build-stamp.json +3 -3
- package/dist/commands/init.js +13 -9
- package/dist/commands/task-advance.js +32 -0
- package/dist/executors/dag-pi-executor.js +1615 -188
- package/dist/executors/pi-executor.js +85 -3
- package/dist/executors/pi-sdk-executor.js +18 -0
- package/dist/executors/shell-executor.js +92 -7
- package/dist/shared/backend-dogfood-preflight.js +47 -0
- package/dist/shared/dag-failure-category.js +3 -0
- package/dist/shared/operator/capabilities.js +180 -0
- package/dist/task/config-types.js +29 -0
- package/dist/task/contract/project.js +9 -0
- package/dist/task/contract/schema.js +17 -0
- package/dist/task/source-prepare/build-draft.js +60 -0
- package/dist/worker/console/chat/pi-mode-loop-isolation.js +155 -0
- package/dist/worker/console/chat/pi-runtime/custom-tools/operator-tools.js +24 -16
- package/dist/worker/console/chat/pi-runtime/custom-tools/scheduled-goal-tools.js +22 -0
- package/dist/worker/console/chat/pi-runtime/progressive-tools.js +195 -0
- package/dist/worker/console/chat/pi-runtime.js +86 -27
- package/dist/worker/console/chat/routes.js +19 -0
- package/dist/worker/console/chat/scheduled-goal-booking.js +59 -0
- package/dist/worker/console/chat/scheduled-goal-delivery.js +27 -0
- package/dist/worker/console/chat/scheduled-goal-request.js +190 -0
- package/dist/worker/console/chat/session-mode.js +7 -3
- package/dist/worker/console/chat/session-store.js +42 -9
- package/dist/worker/console/chat/tool-preview.js +115 -0
- package/dist/worker/console/chat/turn-order.js +13 -0
- package/dist/worker/console/chat/turn-process.js +32 -30
- package/dist/worker/console/operator-actions.js +50 -0
- package/dist/worker/console/prd-intake-bridge.js +54 -1
- package/dist/worker/console/scheduled-goal-host.js +98 -0
- package/dist/worker/console/scheduled-goal-operation-adapter.js +127 -0
- package/dist/worker/console/server.js +15 -0
- package/dist/worker/console/static/assets/{abnfDiagram-N423BO3Z-B_UDknhR.js → abnfDiagram-N423BO3Z-8-j6y-sd.js} +1 -1
- package/dist/worker/console/static/assets/{arc-D6hU0drN.js → arc-eoQiMvuk.js} +1 -1
- package/dist/worker/console/static/assets/{architectureDiagram-T3A2C74G-DmVikdRK.js → architectureDiagram-T3A2C74G-C4A3uMcI.js} +1 -1
- package/dist/worker/console/static/assets/{blockDiagram-VBNYF7ZC-DydABCRg.js → blockDiagram-VBNYF7ZC-D4zD2F-Q.js} +1 -1
- package/dist/worker/console/static/assets/{c4Diagram-5PPSVZJV-DrXKC2Jl.js → c4Diagram-5PPSVZJV-j1RkJziL.js} +1 -1
- package/dist/worker/console/static/assets/channel-ChE7y-cx.js +1 -0
- package/dist/worker/console/static/assets/{chunk-2GRJ4B5K-D2-qFxr7.js → chunk-2GRJ4B5K-YBHmhik1.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-2Q5K7J3B-ChfeF41P.js → chunk-2Q5K7J3B-COfWyo9P.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-5RXB4S5H-Drwrz55x.js → chunk-5RXB4S5H-I99OUkHY.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-5VM5RSS4-CAQGErL7.js → chunk-5VM5RSS4-XKxoJNJ4.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-6Q2QTUOP-DHO38w5U.js → chunk-6Q2QTUOP-DLT_cYx1.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-GF5L2VYU-EYIN0EOI.js → chunk-GF5L2VYU-CxcKZ9mV.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-JWPE2WC7-CrFLKqkT.js → chunk-JWPE2WC7-V1EqXdjY.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-KBJHAD2P-DMchffoo.js → chunk-KBJHAD2P-DebTtdFp.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-RYQCIY6F-DaSvtKQa.js → chunk-RYQCIY6F-B-ivNRDf.js} +1 -1
- package/dist/worker/console/static/assets/{chunk-XXDRQBXY-DcW8W9Qj.js → chunk-XXDRQBXY-DC_Ds11b.js} +1 -1
- package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-C01TCf2X.js +1 -0
- package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-C01TCf2X.js +1 -0
- package/dist/worker/console/static/assets/{cose-bilkent-JH36ORCC-BUirrMEi.js → cose-bilkent-JH36ORCC-sWEqKwIb.js} +1 -1
- package/dist/worker/console/static/assets/{cynefin-VYW2F7L2-BXzO8iXR.js → cynefin-VYW2F7L2-BXa_dcu4.js} +1 -1
- package/dist/worker/console/static/assets/{cynefinDiagram-MW4NZA55-DAtgaxIn.js → cynefinDiagram-MW4NZA55-C-Mle84F.js} +1 -1
- package/dist/worker/console/static/assets/{dagre-VZM6K2ZE-C1re9Noh.js → dagre-VZM6K2ZE-CXPDBITe.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-7IWD3JNH-N8Kj08J9.js → diagram-7IWD3JNH-CC-WJQfa.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-B4RE2ZJO-D6BpSLs3.js → diagram-B4RE2ZJO-CDAV5Vs4.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-LBJQPF4R-C2Vlgnl4.js → diagram-LBJQPF4R-CmFiAcNz.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-Q27KOJAE-CMbqtVLS.js → diagram-Q27KOJAE-DPBZHuyn.js} +1 -1
- package/dist/worker/console/static/assets/{diagram-UB23O5K3-A770Eb-0.js → diagram-UB23O5K3-jUlm_Ds3.js} +1 -1
- package/dist/worker/console/static/assets/{ebnfDiagram-BXEA7PRR-LV2w_2pT.js → ebnfDiagram-BXEA7PRR-DWhQ3mfY.js} +1 -1
- package/dist/worker/console/static/assets/{erDiagram-JOGREHBK-bwqf56ah.js → erDiagram-JOGREHBK-Tr2gMqet.js} +1 -1
- package/dist/worker/console/static/assets/{flowDiagram-UKHOOZJN-BnVtHhZh.js → flowDiagram-UKHOOZJN-DJjVQHPA.js} +1 -1
- package/dist/worker/console/static/assets/{ganttDiagram-PKOTCBZU-BGVToEqc.js → ganttDiagram-PKOTCBZU-D74dQ4u0.js} +1 -1
- package/dist/worker/console/static/assets/{gitGraphDiagram-DS77QQ5N-DL2l7Vne.js → gitGraphDiagram-DS77QQ5N-DwW0tZ0X.js} +1 -1
- package/dist/worker/console/static/assets/index-BWkIfcrK.css +1 -0
- package/dist/worker/console/static/assets/{index-B_V4wvXs.js → index-Cdkvw_H6.js} +97 -97
- package/dist/worker/console/static/assets/{infoDiagram-6WML65LV-BWIpubOx.js → infoDiagram-6WML65LV-ILCbxJyb.js} +1 -1
- package/dist/worker/console/static/assets/{ishikawaDiagram-WSZJBQD7-DjYBD8vv.js → ishikawaDiagram-WSZJBQD7-DeuWBQ89.js} +1 -1
- package/dist/worker/console/static/assets/{journeyDiagram-NVQOT4AX-C0_mBaHb.js → journeyDiagram-NVQOT4AX-CF5ih8Fk.js} +1 -1
- package/dist/worker/console/static/assets/{kanban-definition-27J2QSJJ-Cy71zbX6.js → kanban-definition-27J2QSJJ-C7yOSuRO.js} +1 -1
- package/dist/worker/console/static/assets/{linear-Dt3_w3Vn.js → linear-BDZ9riWi.js} +1 -1
- package/dist/worker/console/static/assets/{mermaid.core-DnNOlfEP.js → mermaid.core-7pKqYtpZ.js} +5 -5
- package/dist/worker/console/static/assets/{mindmap-definition-FAOFIHXS-DAbTspwq.js → mindmap-definition-FAOFIHXS-LeJDybSU.js} +1 -1
- package/dist/worker/console/static/assets/{pegDiagram-VL7TDLO6-NNiM141p.js → pegDiagram-VL7TDLO6-BJvT3pMD.js} +1 -1
- package/dist/worker/console/static/assets/{pieDiagram-7S7Q4E2Y-CN80Z1Do.js → pieDiagram-7S7Q4E2Y-_rGqpMan.js} +1 -1
- package/dist/worker/console/static/assets/{quadrantDiagram-CIZ2JOQS-Bxyfh0JU.js → quadrantDiagram-CIZ2JOQS-DPcSv5aA.js} +1 -1
- package/dist/worker/console/static/assets/{railroadDiagram-AXF67PYL-BlWnC79M.js → railroadDiagram-AXF67PYL-Drxx4hkJ.js} +1 -1
- package/dist/worker/console/static/assets/{requirementDiagram-LRYGKXZP-BrEu4P_e.js → requirementDiagram-LRYGKXZP-BCTUdU4z.js} +1 -1
- package/dist/worker/console/static/assets/{sankeyDiagram-W5VNT64P-Bt70igBr.js → sankeyDiagram-W5VNT64P-B6wzbZmB.js} +1 -1
- package/dist/worker/console/static/assets/{sequenceDiagram-SI44F4Z6-C8DQR-r9.js → sequenceDiagram-SI44F4Z6-BGt8d3QQ.js} +1 -1
- package/dist/worker/console/static/assets/{sizeCapture-X5ZJPWSS-DEMinJAT.js → sizeCapture-X5ZJPWSS-7nlhJKo7.js} +1 -1
- package/dist/worker/console/static/assets/{stateDiagram-OKZ733FA-DDdDBzQc.js → stateDiagram-OKZ733FA-Cv2stqMA.js} +1 -1
- package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-DC7V5vcq.js +1 -0
- package/dist/worker/console/static/assets/{swimlanes-SLNWSIFB-G2wcBs3T.js → swimlanes-SLNWSIFB-qDAo4Yc1.js} +2 -2
- package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-Bwy4QUTO.js +8 -0
- package/dist/worker/console/static/assets/{timeline-definition-Z64GVDOM-Dzn4g2gJ.js → timeline-definition-Z64GVDOM-CKbDNKTr.js} +1 -1
- package/dist/worker/console/static/assets/{vennDiagram-T6HMQDX7-5v2rP9oO.js → vennDiagram-T6HMQDX7-DI-9EHic.js} +1 -1
- package/dist/worker/console/static/assets/{wardleyDiagram-T6FBY63Y-B4EOx7ew.js → wardleyDiagram-T6FBY63Y-BrzzRDpX.js} +1 -1
- package/dist/worker/console/static/assets/{xychartDiagram-ELKLHX3M-CQ1v8eYP.js → xychartDiagram-ELKLHX3M-BclH5hGh.js} +1 -1
- package/dist/worker/console/static/index.html +2 -2
- package/dist/worker/console/static-src/operator-chat/chat-sse-events.js +41 -11
- package/dist/worker/console/static-src/operator-chat/compaction-message.js +3 -17
- package/dist/worker/console/static-src/operator-chat/timeline-merge.js +29 -0
- package/dist/worker/console/static-src/operator-chat/useChatThread.js +2 -5
- package/dist/worker/console/workspace-context.js +11 -0
- package/dist/worker/observe/static/operator-chrome.js +3 -1
- package/dist/worker/scheduler/scheduled-goal-dispatch.js +117 -0
- package/dist/worker/scheduler/scheduled-goal-evidence.js +167 -0
- package/dist/worker/scheduler/scheduled-goal-recovery.js +62 -0
- package/dist/worker/scheduler/scheduled-goal-store.js +699 -0
- package/dist/worker/scheduler/scheduled-goal-supervisor.js +132 -0
- package/dist/worker/scheduler/scheduled-goal-time.js +102 -0
- package/dist/worker/scheduler/scheduled-goal-types.js +95 -0
- package/dist/workflows/dag/backend-test-markdown-workflow.js +6 -24
- package/dist/workflows/dag/backend-test-plan-protocol.js +82 -5
- package/dist/workflows/dag/dag-retry-schema.js +11 -0
- package/dist/workflows/dag/frontend-closeout.js +4 -2
- package/dist/workflows/dag/frontend-committed-facts.js +461 -0
- package/dist/workflows/dag/frontend-durable-tools.js +15 -3
- package/dist/workflows/dag/frontend-implementation-contract.js +365 -8
- package/dist/workflows/dag/frontend-plan-canary.js +53 -0
- package/dist/workflows/dag/frontend-plan-decision-contract.js +803 -0
- package/dist/workflows/dag/frontend-plan-render.js +0 -2
- package/dist/workflows/dag/frontend-prewrite-gate.js +3 -3
- package/dist/workflows/dag/frontend-provider-capability-matrix.js +8 -61
- package/dist/workflows/dag/frontend-recovery-run.js +57 -0
- package/dist/workflows/dag/frontend-review-context.js +43 -70
- package/dist/workflows/dag/frontend-risk.js +92 -10
- package/dist/workflows/dag/frontend-session-budget.js +117 -3
- package/dist/workflows/dag/frontend-shape.js +11 -55
- package/dist/workflows/dag/frontend-test-execution-evidence.js +35 -10
- package/dist/workflows/dag/frontend-typed-event-store.js +32 -25
- package/dist/workflows/dag/frontend-verification-trace.js +11 -2
- package/dist/workflows/dag/frontend-writer-admission.js +3 -47
- package/dist/workflows/dag/frontend-writer-status.js +0 -23
- package/dist/workflows/dag/init-hybrid.js +182 -185
- package/dist/workflows/dag/lifecycle.js +7 -2
- package/dist/workflows/dag/node-execution.js +48 -13
- package/dist/workflows/dag/rerun-plan.js +10 -0
- package/dist/workflows/dag/rerun-task.js +77 -1
- package/dist/workflows/dag/retry-policy.js +18 -0
- package/dist/workflows/dag/scheduler.js +2 -4
- package/dist/workflows/dag/types.js +23 -0
- package/docs/README.md +1 -0
- package/docs/init-surface.manifest.json +1 -0
- package/docs/operations/README.md +2 -0
- package/docs/templates/README.md +1 -1
- package/docs/templates/agent-dag.schema.json +2 -2
- package/docs/templates/backend-test-dag.json +14 -10
- package/docs/templates/frontend-implementation-contract.schema.json +0 -7
- package/docs/templates/frontend-implementation-dag.json +8 -9
- package/package.json +6 -3
- package/skills/frontend-bounded-implement/SKILL.md +3 -4
- package/skills/frontend-design-review/SKILL.md +7 -12
- package/skills/frontend-design-review/references/review-checklist.md +7 -7
- package/skills/frontend-plan/SKILL.md +8 -18
- package/skills/frontend-plan/references/decision-contract.md +19 -26
- package/skills/frontend-review/SKILL.md +21 -25
- package/skills/frontend-review/references/review-findings.md +48 -24
- package/dist/worker/console/static/assets/channel-Drecd94a.js +0 -1
- package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-8Udu0t8-.js +0 -1
- package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-8Udu0t8-.js +0 -1
- package/dist/worker/console/static/assets/index-DcudonhZ.css +0 -1
- package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-CyCiS0Lt.js +0 -1
- package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-U0lhi6N0.js +0 -8
- package/dist/workflows/dag/frontend-shadow-dual-write.js +0 -975
|
@@ -97,8 +97,6 @@ export function renderFrontendPlanMarkdown(contract) {
|
|
|
97
97
|
"## Source Fidelity Ledger",
|
|
98
98
|
sourceFidelityLedger(contract),
|
|
99
99
|
"",
|
|
100
|
-
"## Implementation Steps",
|
|
101
|
-
bulletList(contract.implementationSteps) || "_(not specified)_",
|
|
102
100
|
"",
|
|
103
101
|
"## Target Files",
|
|
104
102
|
bulletList(contract.targets.files) || "_(none)_",
|
|
@@ -702,10 +702,10 @@ export async function runFrontendPrewriteGate(input) {
|
|
|
702
702
|
failureCode: "mock-strategy-outside-allowed",
|
|
703
703
|
});
|
|
704
704
|
}
|
|
705
|
-
//
|
|
706
|
-
//
|
|
705
|
+
// Mock command binding (main): a non-not-needed strategy can only be proven
|
|
706
|
+
// by the DAG's frozen Mock verification commands. Block before any write is
|
|
707
|
+
// authorized when none were materialized at generation time.
|
|
707
708
|
if (mockStrategy !== "not-needed" &&
|
|
708
|
-
analysis.canonical.verificationTargets.some((target) => target.mode === "behavior") &&
|
|
709
709
|
(input.config.mockCommandLabels?.length ?? 0) === 0) {
|
|
710
710
|
return finalizePrewrite(input, {
|
|
711
711
|
...basePending,
|
|
@@ -1,8 +1,7 @@
|
|
|
1
1
|
import { randomUUID } from "node:crypto";
|
|
2
2
|
import { TYPED_EVENT_FACT_KINDS, stageTypedEventRecord, } from "./frontend-typed-event-store.js";
|
|
3
|
-
import { createDagEventObserver } from "./event-observer.js";
|
|
4
3
|
/**
|
|
5
|
-
*
|
|
4
|
+
* Provider/model capability fact matrix (AC-003).
|
|
6
5
|
*
|
|
7
6
|
* Records seven provider/model capability facts as read-only observations:
|
|
8
7
|
* - `terminal-commit`
|
|
@@ -13,15 +12,12 @@ import { createDagEventObserver } from "./event-observer.js";
|
|
|
13
12
|
* - `provider-ended-without-terminal`
|
|
14
13
|
* - `read-only-fact-channel`
|
|
15
14
|
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
* control path. It is deliberately distinct from `frontend-project-capability`
|
|
20
|
-
* (framework / mock discovery).
|
|
15
|
+
* `recordProviderCapabilityFact` returns a staged record for audit, never a
|
|
16
|
+
* branch-control signal, and this module is deliberately distinct from
|
|
17
|
+
* `frontend-project-capability` (framework / mock discovery).
|
|
21
18
|
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
24
|
-
* failure can never affect canonical DAG execution.
|
|
19
|
+
* These facts are staged independently from the canonical DAG event stream;
|
|
20
|
+
* recording them never participates in model selection or branch control.
|
|
25
21
|
*/
|
|
26
22
|
export const PROVIDER_CAPABILITY_FACT_KINDS = TYPED_EVENT_FACT_KINDS;
|
|
27
23
|
/** Terminal finish-reason shapes, modeled after
|
|
@@ -55,8 +51,8 @@ function readExplicitFinishReason(source) {
|
|
|
55
51
|
function isRecord(value) {
|
|
56
52
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
57
53
|
}
|
|
58
|
-
/** True when an observed provider event is terminal-shaped.
|
|
59
|
-
*
|
|
54
|
+
/** True when an observed provider event is terminal-shaped. The result is a
|
|
55
|
+
* fact, not a control decision. */
|
|
60
56
|
export function isTerminalProviderEventShape(event) {
|
|
61
57
|
const type = typeof event.type === "string" ? event.type : undefined;
|
|
62
58
|
return type !== undefined && TERMINAL_FINISH_REASON_TYPES.has(type);
|
|
@@ -108,52 +104,3 @@ export function recordProviderCapabilityFact(options) {
|
|
|
108
104
|
});
|
|
109
105
|
return { eventId, record };
|
|
110
106
|
}
|
|
111
|
-
/**
|
|
112
|
-
* Attach a shadow observer that records `read-only-fact-channel` facts while
|
|
113
|
-
* delegating the canonical event JSONL to the existing `createDagEventObserver`.
|
|
114
|
-
* The shadow channel never throws into the caller: any recording failure is
|
|
115
|
-
* swallowed so canonical execution is unaffected.
|
|
116
|
-
*/
|
|
117
|
-
export function createProviderCapabilityShadowObserver(options) {
|
|
118
|
-
const base = createDagEventObserver({
|
|
119
|
-
eventsJsonlPath: options.eventsJsonlPath,
|
|
120
|
-
dagRunId: options.dagRunId,
|
|
121
|
-
spec: options.spec,
|
|
122
|
-
});
|
|
123
|
-
const recordReadOnlyFact = (eventType) => {
|
|
124
|
-
try {
|
|
125
|
-
recordProviderCapabilityFact({
|
|
126
|
-
store: options.store,
|
|
127
|
-
attemptId: options.attemptId,
|
|
128
|
-
requestId: options.requestId,
|
|
129
|
-
fact: { kind: "read-only-fact-channel", eventType },
|
|
130
|
-
});
|
|
131
|
-
}
|
|
132
|
-
catch {
|
|
133
|
-
// Shadow channel must never affect canonical execution.
|
|
134
|
-
}
|
|
135
|
-
};
|
|
136
|
-
const observer = {
|
|
137
|
-
onRunStart: async (state) => {
|
|
138
|
-
recordReadOnlyFact("dag.run.started");
|
|
139
|
-
await base.observer.onRunStart?.(state);
|
|
140
|
-
},
|
|
141
|
-
onNodeStart: async (nodeId, state) => {
|
|
142
|
-
recordReadOnlyFact("dag.node.started");
|
|
143
|
-
await base.observer.onNodeStart?.(nodeId, state);
|
|
144
|
-
},
|
|
145
|
-
onNodeOutput: async (nodeId, chunk, state) => {
|
|
146
|
-
recordReadOnlyFact("dag.node.output");
|
|
147
|
-
await base.observer.onNodeOutput?.(nodeId, chunk, state);
|
|
148
|
-
},
|
|
149
|
-
onNodeFinish: async (nodeId, state) => {
|
|
150
|
-
recordReadOnlyFact("dag.node.finished");
|
|
151
|
-
await base.observer.onNodeFinish?.(nodeId, state);
|
|
152
|
-
},
|
|
153
|
-
onRunFinish: async (state) => {
|
|
154
|
-
recordReadOnlyFact("dag.run.finished");
|
|
155
|
-
await base.observer.onRunFinish?.(state);
|
|
156
|
-
},
|
|
157
|
-
};
|
|
158
|
-
return { observer, flush: base.flush };
|
|
159
|
-
}
|
|
@@ -25,6 +25,27 @@ import { topoSortToRanks } from "./topo.js";
|
|
|
25
25
|
/** Canonical contract artifact default location; overridden by the design
|
|
26
26
|
* policy shell's configured outputDir/artifactName when present. */
|
|
27
27
|
const FRONTEND_CANONICAL_CONTRACT_DEFAULT_REL_PATH = "contracts/frontend-implementation-contract.json";
|
|
28
|
+
/**
|
|
29
|
+
* Run-owned artifacts a frontend node writes *outside* its own node directory.
|
|
30
|
+
*
|
|
31
|
+
* An imported node keeps its parent FINISHED record in the child run, so the
|
|
32
|
+
* child never executes the body that produced these files. Any downstream node
|
|
33
|
+
* that reads them must still find them in the child run directory: smoke r31
|
|
34
|
+
* saw a review-triggered recovery die at `frontend-review-context-shell` with
|
|
35
|
+
* `ENOENT contracts/frontend-worktree-baseline.json`, because
|
|
36
|
+
* `frontend-writer-admission-shell` is imported (its baseline capture is a side
|
|
37
|
+
* effect of the node body) while the review context requires that file.
|
|
38
|
+
*
|
|
39
|
+
* Entries listed as required fail the staging closed when the parent lacks
|
|
40
|
+
* them; optional entries (for example the lint baseline, which only exists when
|
|
41
|
+
* lint is configured) are copied when present.
|
|
42
|
+
*/
|
|
43
|
+
const FRONTEND_IMPORTED_NODE_RUN_ARTIFACTS = {
|
|
44
|
+
"frontend-writer-admission-shell": {
|
|
45
|
+
required: ["contracts/frontend-worktree-baseline.json"],
|
|
46
|
+
optional: ["contracts/frontend-lint-baseline.json"],
|
|
47
|
+
},
|
|
48
|
+
};
|
|
28
49
|
function resolveDeps(deps) {
|
|
29
50
|
return {
|
|
30
51
|
readParentState: deps?.readParentState ?? readDagRunState,
|
|
@@ -397,6 +418,41 @@ async function materializeChildStaging(input) {
|
|
|
397
418
|
newRunDir: stagingDir,
|
|
398
419
|
relativePath: FRONTEND_WRITER_ADMISSION_RESULT_ARTIFACT,
|
|
399
420
|
});
|
|
421
|
+
// Imported nodes keep their parent FINISHED record, so the run-owned files
|
|
422
|
+
// their bodies wrote outside the node directory travel with them (smoke r31:
|
|
423
|
+
// a review-triggered recovery reached review-context with no worktree
|
|
424
|
+
// baseline and hard-failed on ENOENT).
|
|
425
|
+
const importedRunArtifacts = [];
|
|
426
|
+
for (const nodeId of plan.importedNodeIds) {
|
|
427
|
+
const artifactSpec = FRONTEND_IMPORTED_NODE_RUN_ARTIFACTS[nodeId];
|
|
428
|
+
if (!artifactSpec)
|
|
429
|
+
continue;
|
|
430
|
+
for (const relativePath of artifactSpec.required) {
|
|
431
|
+
const copied = await copyArtifactVerified({
|
|
432
|
+
parentRunDir,
|
|
433
|
+
newRunDir: stagingDir,
|
|
434
|
+
relativePath,
|
|
435
|
+
});
|
|
436
|
+
importedRunArtifacts.push({
|
|
437
|
+
sourceRelativePath: relativePath,
|
|
438
|
+
...copied,
|
|
439
|
+
});
|
|
440
|
+
}
|
|
441
|
+
for (const relativePath of artifactSpec.optional) {
|
|
442
|
+
const sourcePath = path.join(parentRunDir, ...relativePath.split("/"));
|
|
443
|
+
if (!(await fileExists(sourcePath)))
|
|
444
|
+
continue;
|
|
445
|
+
const copied = await copyArtifactVerified({
|
|
446
|
+
parentRunDir,
|
|
447
|
+
newRunDir: stagingDir,
|
|
448
|
+
relativePath,
|
|
449
|
+
});
|
|
450
|
+
importedRunArtifacts.push({
|
|
451
|
+
sourceRelativePath: relativePath,
|
|
452
|
+
...copied,
|
|
453
|
+
});
|
|
454
|
+
}
|
|
455
|
+
}
|
|
400
456
|
// The scheduler gates every shape boundary by its committed fact once the
|
|
401
457
|
// boundary's source node is FINISHED. Imported nodes stay FINISHED in the
|
|
402
458
|
// child, so their boundaries must inherit the parent's committed facts or
|
|
@@ -470,6 +526,7 @@ async function materializeChildStaging(input) {
|
|
|
470
526
|
relativePath: writerAdmissionResult.destinationRelativePath,
|
|
471
527
|
sha256: writerAdmissionResult.sha256,
|
|
472
528
|
},
|
|
529
|
+
importedRunArtifacts,
|
|
473
530
|
importedFacts,
|
|
474
531
|
};
|
|
475
532
|
await writeJsonVerified(stagingDir, FRONTEND_RECOVERY_IMPORT_MANIFEST_REL_PATH, importManifest);
|
|
@@ -96,74 +96,6 @@ export async function captureCumulativeDiffContext(input) {
|
|
|
96
96
|
}
|
|
97
97
|
return assembleCumulativeDiffContext({ baselineEntries, currentEntries });
|
|
98
98
|
}
|
|
99
|
-
const TYPED_TO_LEGACY = {
|
|
100
|
-
approve_review: "pass",
|
|
101
|
-
request_review_changes: "request-revision",
|
|
102
|
-
};
|
|
103
|
-
function isTypedReviewTerminalVerdict(value) {
|
|
104
|
-
return value === "approve_review" || value === "request_review_changes";
|
|
105
|
-
}
|
|
106
|
-
function isLegacyReviewVerdict(value) {
|
|
107
|
-
return value === "pass" || value === "request-revision";
|
|
108
|
-
}
|
|
109
|
-
/**
|
|
110
|
-
* Deterministic, fail-closed equivalence between the typed review terminal
|
|
111
|
-
* verdict(s) observed in the attempt and the legacy JSON verdict.
|
|
112
|
-
*
|
|
113
|
-
* Fail-closed cases (all yield `match: false` with an explicit reason):
|
|
114
|
-
* - no typed terminal fact;
|
|
115
|
-
* - more than one typed terminal fact (conflicting);
|
|
116
|
-
* - an unknown typed terminal kind;
|
|
117
|
-
* - a missing/illegal legacy JSON verdict.
|
|
118
|
-
*/
|
|
119
|
-
export function compareTypedReviewToLegacyJsonVerdict(input) {
|
|
120
|
-
const kinds = [...input.typedKinds];
|
|
121
|
-
const legacyVerdict = isLegacyReviewVerdict(input.legacyVerdict ?? "")
|
|
122
|
-
? input.legacyVerdict
|
|
123
|
-
: undefined;
|
|
124
|
-
if (kinds.length === 0) {
|
|
125
|
-
return {
|
|
126
|
-
typedVerdict: undefined,
|
|
127
|
-
legacyVerdict,
|
|
128
|
-
match: false,
|
|
129
|
-
reason: "typed review terminal fact missing",
|
|
130
|
-
};
|
|
131
|
-
}
|
|
132
|
-
if (kinds.length > 1) {
|
|
133
|
-
return {
|
|
134
|
-
typedVerdict: undefined,
|
|
135
|
-
legacyVerdict,
|
|
136
|
-
match: false,
|
|
137
|
-
reason: `conflicting typed review terminal facts: ${[...new Set(kinds)].join(",")}`,
|
|
138
|
-
};
|
|
139
|
-
}
|
|
140
|
-
const typedVerdict = kinds[0];
|
|
141
|
-
if (!isTypedReviewTerminalVerdict(typedVerdict)) {
|
|
142
|
-
return {
|
|
143
|
-
typedVerdict: undefined,
|
|
144
|
-
legacyVerdict,
|
|
145
|
-
match: false,
|
|
146
|
-
reason: `unknown typed review terminal fact kind: ${typedVerdict}`,
|
|
147
|
-
};
|
|
148
|
-
}
|
|
149
|
-
if (!legacyVerdict) {
|
|
150
|
-
return {
|
|
151
|
-
typedVerdict,
|
|
152
|
-
legacyVerdict: undefined,
|
|
153
|
-
match: false,
|
|
154
|
-
reason: "legacy JSON verdict missing or not pass|request-revision",
|
|
155
|
-
};
|
|
156
|
-
}
|
|
157
|
-
const expected = TYPED_TO_LEGACY[typedVerdict];
|
|
158
|
-
return {
|
|
159
|
-
typedVerdict,
|
|
160
|
-
legacyVerdict,
|
|
161
|
-
match: expected === legacyVerdict,
|
|
162
|
-
reason: expected === legacyVerdict
|
|
163
|
-
? "typed review verdict matches legacy JSON verdict"
|
|
164
|
-
: `typed review verdict ${typedVerdict} maps to ${expected} but legacy JSON verdict is ${legacyVerdict}`,
|
|
165
|
-
};
|
|
166
|
-
}
|
|
167
99
|
/**
|
|
168
100
|
* review-context 稳定失败枚举(AC-011 / AC-HARD-005):v2 drift / manifest 失败返回
|
|
169
101
|
* 结构化、可消费的 failure code(稳定枚举 + reason 的 envelope),既有调用方无需字符串
|
|
@@ -191,6 +123,37 @@ async function readRequiredJson(runDir, relativePath) {
|
|
|
191
123
|
throw new FrontendReviewContextFailure("review-context-manifest-missing", `frontend review context missing or invalid ${relativePath}: ${error instanceof Error ? error.message : String(error)}`);
|
|
192
124
|
}
|
|
193
125
|
}
|
|
126
|
+
/**
|
|
127
|
+
* Per-command verification evidence written by the verify shell.
|
|
128
|
+
*
|
|
129
|
+
* The review protocol treats shell exit status as authoritative ("typecheck,
|
|
130
|
+
* build, and test are required successful final exits"), but the verification
|
|
131
|
+
* trace only proves command/file/target-id *binding*. Smoke r31: with no exit
|
|
132
|
+
* evidence in the context, a compliant reviewer had to request changes for a
|
|
133
|
+
* missing check it could not perform. Absent the artifact the context says so
|
|
134
|
+
* explicitly instead of leaving the reviewer to guess.
|
|
135
|
+
*/
|
|
136
|
+
async function readVerificationEvidence(runDir) {
|
|
137
|
+
const relativePath = "contracts/frontend-verification-evidence.json";
|
|
138
|
+
try {
|
|
139
|
+
const decoded = JSON.parse(await readFile(path.join(runDir, relativePath), "utf8"));
|
|
140
|
+
if (decoded?.schemaId !== "frontend-verification-evidence-v1") {
|
|
141
|
+
return {
|
|
142
|
+
schemaId: "frontend-verification-evidence-v1",
|
|
143
|
+
status: "unavailable",
|
|
144
|
+
reason: `${relativePath} carries an unexpected schemaId`,
|
|
145
|
+
};
|
|
146
|
+
}
|
|
147
|
+
return { ...decoded, status: "available" };
|
|
148
|
+
}
|
|
149
|
+
catch (error) {
|
|
150
|
+
return {
|
|
151
|
+
schemaId: "frontend-verification-evidence-v1",
|
|
152
|
+
status: "unavailable",
|
|
153
|
+
reason: `${relativePath} unavailable: ${error instanceof Error ? error.message : String(error)}`,
|
|
154
|
+
};
|
|
155
|
+
}
|
|
156
|
+
}
|
|
194
157
|
async function readOptionalLintAssessment(runDir) {
|
|
195
158
|
const relativePath = "contracts/frontend-lint-assessment.json";
|
|
196
159
|
try {
|
|
@@ -246,8 +209,7 @@ function assertReviewEvidence(input) {
|
|
|
246
209
|
const trace = input.verificationTrace;
|
|
247
210
|
const expectedTargets = new Map(contract.data.verificationTargets.map((target) => [target.id, target]));
|
|
248
211
|
const actualTargets = Array.isArray(trace?.targets) ? trace.targets : [];
|
|
249
|
-
const mockEvidenceBound = contract.data.
|
|
250
|
-
contract.data.mockApi.strategy === "not-needed" ||
|
|
212
|
+
const mockEvidenceBound = contract.data.mockApi.strategy === "not-needed" ||
|
|
251
213
|
(typeof trace?.mockCommandNodeId === "string" && trace.mockCommandNodeId.length > 0 &&
|
|
252
214
|
Array.isArray(trace?.mockCommandLabels) && trace.mockCommandLabels.length > 0 &&
|
|
253
215
|
Array.isArray(trace?.mockCommandTexts) && trace.mockCommandTexts.length === trace.mockCommandLabels.length &&
|
|
@@ -321,6 +283,7 @@ export async function runFrontendReviewContextGate(input) {
|
|
|
321
283
|
const repairAssessment = await readOptionalRepairAssessment(input.runDir);
|
|
322
284
|
const verifyFailure = await readOptionalVerifyFailure(input.runDir);
|
|
323
285
|
const lintAssessment = await readOptionalLintAssessment(input.runDir);
|
|
286
|
+
const verificationEvidence = await readVerificationEvidence(input.runDir);
|
|
324
287
|
const cumulativeDiff = await captureCumulativeDiffContext({
|
|
325
288
|
runDir: input.runDir,
|
|
326
289
|
workspaceRoot: input.workspaceRoot,
|
|
@@ -356,6 +319,16 @@ export async function runFrontendReviewContextGate(input) {
|
|
|
356
319
|
sections: contractReviewArtifacts.index.sections,
|
|
357
320
|
},
|
|
358
321
|
verificationTrace,
|
|
322
|
+
verificationEvidence,
|
|
323
|
+
// Protocol-legal lint vocabulary; `lintConfigured: false` tells the
|
|
324
|
+
// reviewer that this task declares no lint surface at all, so
|
|
325
|
+
// `unavailable` reads as "not applicable", not "missing evidence".
|
|
326
|
+
lintStatus: lintAssessment
|
|
327
|
+
? lintAssessment.status
|
|
328
|
+
: typeof verificationEvidence.lintStatus === "string"
|
|
329
|
+
? verificationEvidence.lintStatus
|
|
330
|
+
: "unavailable",
|
|
331
|
+
lintConfigured: verificationEvidence.lintConfigured === true,
|
|
359
332
|
...(repairAssessment ? { repairAssessment } : {}),
|
|
360
333
|
...(verifyFailure ? { verifyFailure } : {}),
|
|
361
334
|
...(lintAssessment ? { lintAssessment } : {}),
|
|
@@ -7,8 +7,10 @@
|
|
|
7
7
|
* allowing its path is NOT dependency introduction by itself; the manifest
|
|
8
8
|
* path check in classifyFrontendRisk requires this language to escalate.
|
|
9
9
|
*/
|
|
10
|
-
const DEPENDENCY_INSTALL_LANGUAGE = /new\s+dependency|add\s+
|
|
11
|
-
|
|
10
|
+
const DEPENDENCY_INSTALL_LANGUAGE = /new\s+dependency|add\s+dependenc(?:y|ies)|install\s+dependenc(?:y|ies)|npm\s+(?:install|i)\b|yarn\s+(?:add|install)\b|pnpm\s+(?:add|install)\b|bun\s+(?:add|install)\b|新增依赖|添加依赖|安装依赖|引入依赖/i;
|
|
11
|
+
/** Shared vocabulary for risk and topology routing. Keep policy-specific
|
|
12
|
+
* signals (design conflict/state infrastructure) below this list. */
|
|
13
|
+
export const FRONTEND_SHARED_ESCALATION_PATTERNS = [
|
|
12
14
|
{
|
|
13
15
|
id: "new-route",
|
|
14
16
|
re: /new\s+route|新增路由|createBrowserRouter|add\s+route|新建页面路由/i,
|
|
@@ -45,6 +47,48 @@ const HIGH_RISK_PATTERNS = [
|
|
|
45
47
|
id: "cross-domain-paths",
|
|
46
48
|
re: /多个领域|cross-module|multiple\s+packages/i,
|
|
47
49
|
},
|
|
50
|
+
];
|
|
51
|
+
function isExplicitSharedExclusion(clause, pattern) {
|
|
52
|
+
const match = new RegExp(pattern.re.source, "i").exec(clause);
|
|
53
|
+
if (!match)
|
|
54
|
+
return false;
|
|
55
|
+
const prefix = clause.slice(0, match.index).trim();
|
|
56
|
+
const suffix = clause.slice(match.index + match[0].length).trim();
|
|
57
|
+
const actionInSignal = ["new-route", "new-dependency"].includes(pattern.id);
|
|
58
|
+
if (suffix && !(actionInSignal && /^no$/i.test(prefix) && /^is\s+(?:required|needed)$/i.test(suffix)))
|
|
59
|
+
return false;
|
|
60
|
+
if (/^(?:不涉及|不(?:修改|变更|调整|改变|新增|添加)|无需(?:修改|变更|调整|改变|新增|添加)|no\s+changes?\s+to|(?:do\s+not|don't)\s+(?:change|modify))$/i.test(prefix))
|
|
61
|
+
return true;
|
|
62
|
+
return actionInSignal && /^(?:不|禁止|无需|无须|不需要|no|without|do\s+not|don't)$/i.test(prefix);
|
|
63
|
+
}
|
|
64
|
+
/** Only a complete, independent scope exclusion can suppress a risk signal.
|
|
65
|
+
* Nested negation, conditions and trailing consequences remain risk evidence;
|
|
66
|
+
* an excluded occurrence never cancels another occurrence elsewhere. */
|
|
67
|
+
export function hasAffirmativeSharedSignal(blob, pattern) {
|
|
68
|
+
for (const sentence of blob.split(/[\n。.;;!?!?]/)) {
|
|
69
|
+
if (!pattern.re.test(sentence))
|
|
70
|
+
continue;
|
|
71
|
+
// Commas can attach a condition to an exclusion. A comma-separated list
|
|
72
|
+
// is independent only when every item is itself an explicit exclusion.
|
|
73
|
+
const clauses = sentence.split(/[,,、]/).map(part => part.trim().replace(/^(?:[-*+]|\d+\))\s+/, ""));
|
|
74
|
+
if (clauses.every(clause => FRONTEND_SHARED_ESCALATION_PATTERNS.some(candidate => isExplicitSharedExclusion(clause, candidate))))
|
|
75
|
+
continue;
|
|
76
|
+
return true;
|
|
77
|
+
}
|
|
78
|
+
return false;
|
|
79
|
+
}
|
|
80
|
+
/** Extract the shared escalation vocabulary once. Shape routing reuses this
|
|
81
|
+
* exact result so risk and topology cannot drift on the same input. */
|
|
82
|
+
export function collectFrontendSharedSignals(blob, supervised = false) {
|
|
83
|
+
const signals = FRONTEND_SHARED_ESCALATION_PATTERNS
|
|
84
|
+
.filter((pattern) => hasAffirmativeSharedSignal(blob, pattern))
|
|
85
|
+
.map((pattern) => pattern.id);
|
|
86
|
+
if (supervised)
|
|
87
|
+
signals.push("supervised-profile");
|
|
88
|
+
return [...new Set(signals)];
|
|
89
|
+
}
|
|
90
|
+
const HIGH_RISK_PATTERNS = [
|
|
91
|
+
...FRONTEND_SHARED_ESCALATION_PATTERNS,
|
|
48
92
|
{
|
|
49
93
|
id: "new-state-infra",
|
|
50
94
|
re: /\bredux\b|\bzustand\b|\bpinia\b|\bjotai\b|\brecoil\b|新状态库|state\s+management\s+introduc/i,
|
|
@@ -77,6 +121,41 @@ function pathSpreadScore(paths) {
|
|
|
77
121
|
.filter(Boolean));
|
|
78
122
|
return tops.size;
|
|
79
123
|
}
|
|
124
|
+
/**
|
|
125
|
+
* Narrow-product-surface test for genuinely small topology. A task qualifies
|
|
126
|
+
* only when its non-auxiliary (product) paths are at most two and all share the
|
|
127
|
+
* same two-segment prefix — `src/components/Button.tsx` + `test/components/*`
|
|
128
|
+
* is narrow, while `src/components/**` + `src/pages/**` spans two product
|
|
129
|
+
* domains and is not. Direct source files (`src/a.tsx`) collapse to their
|
|
130
|
+
* directory so siblings stay narrow. This is the structured replacement for the
|
|
131
|
+
* fragile "single component / local style" keyword gate: it measures the actual
|
|
132
|
+
* delivery surface instead of relying on how the author worded the request.
|
|
133
|
+
*/
|
|
134
|
+
export function isNarrowProductSurface(paths) {
|
|
135
|
+
const auxiliaryRoot = /^(?:test|tests|__tests__|e2e|cypress|spec|specs|docs|documentation)$/i;
|
|
136
|
+
const product = paths
|
|
137
|
+
.map((raw) => raw.replace(/\\/g, "/").split("/").filter(Boolean))
|
|
138
|
+
.filter((segs) => segs.length > 0 && !auxiliaryRoot.test(segs[0] ?? ""));
|
|
139
|
+
if (product.length === 0 || product.length > 2)
|
|
140
|
+
return false;
|
|
141
|
+
// Every product path must name a concrete file or a concrete second-level
|
|
142
|
+
// directory. A top-level wide glob (`src/**`, `src/*`) covers the whole
|
|
143
|
+
// source tree and is NOT a small delivery surface.
|
|
144
|
+
for (const segs of product) {
|
|
145
|
+
const second = segs[1] ?? "";
|
|
146
|
+
if (!second || second === "*" || second === "**")
|
|
147
|
+
return false;
|
|
148
|
+
}
|
|
149
|
+
const prefixOf = (segs) => {
|
|
150
|
+
const second = segs[1] ?? "";
|
|
151
|
+
const secondIsFile = /\.[A-Za-z0-9]+$/.test(second);
|
|
152
|
+
return segs.length >= 2 && second && !secondIsFile
|
|
153
|
+
? `${segs[0] ?? ""}/${segs[1]}`
|
|
154
|
+
: segs[0] ?? "";
|
|
155
|
+
};
|
|
156
|
+
const prefix = prefixOf(product[0] ?? []);
|
|
157
|
+
return product.every((segs) => prefixOf(segs) === prefix);
|
|
158
|
+
}
|
|
80
159
|
/**
|
|
81
160
|
* Classify frontend task risk. High-risk signals always win over small keywords.
|
|
82
161
|
* supervised / explicit high governance never selects small topology.
|
|
@@ -116,10 +195,14 @@ export function classifyFrontendRisk(input) {
|
|
|
116
195
|
if (pattern.re.test(blob))
|
|
117
196
|
smallHits.push(pattern.id);
|
|
118
197
|
}
|
|
119
|
-
// small only if: no high-risk,
|
|
198
|
+
// small only if: no high-risk, non-empty concentrated paths, no remote API
|
|
199
|
+
// signals, not supervised, and not an explicitly large task. Keyword and
|
|
200
|
+
// explicit "small/low" complexity hints are no longer required — they were
|
|
201
|
+
// a fragile proxy for "this is genuinely a small change" and silently
|
|
202
|
+
// forced genuinely-small tasks with different wording into full gates.
|
|
120
203
|
const hasApiSignal = /\b(api|fetch|axios|graphql|endpoint|远程|接口)\b/i.test(blob) &&
|
|
121
204
|
!/\b(no\s+api|无接口|not-needed)\b/i.test(blob);
|
|
122
|
-
const
|
|
205
|
+
const narrowSurface = isNarrowProductSurface(paths);
|
|
123
206
|
if (supervised) {
|
|
124
207
|
rejectedSignals.push(...smallHits.map((id) => `small:${id}:blocked-by-supervised`));
|
|
125
208
|
return {
|
|
@@ -142,19 +225,18 @@ export function classifyFrontendRisk(input) {
|
|
|
142
225
|
forceFullGates: true,
|
|
143
226
|
};
|
|
144
227
|
}
|
|
145
|
-
if (
|
|
146
|
-
concentrated &&
|
|
228
|
+
if (narrowSurface &&
|
|
147
229
|
!hasApiSignal &&
|
|
148
|
-
(input.complexity
|
|
230
|
+
(input.complexity ?? "").toLowerCase() !== "large") {
|
|
149
231
|
return {
|
|
150
232
|
selectedRisk: "small",
|
|
151
|
-
signals: smallHits,
|
|
233
|
+
signals: smallHits.length > 0 ? smallHits : ["structured-small"],
|
|
152
234
|
rejectedSignals,
|
|
153
|
-
reason:
|
|
235
|
+
reason: "narrow single-domain delivery surface without remote API, dependency, or escalation signals; small topology",
|
|
154
236
|
forceFullGates: false,
|
|
155
237
|
};
|
|
156
238
|
}
|
|
157
|
-
if (smallHits.length > 0 && (hasApiSignal || !
|
|
239
|
+
if (smallHits.length > 0 && (hasApiSignal || !narrowSurface)) {
|
|
158
240
|
rejectedSignals.push(...smallHits.map((id) => `small:${id}:blocked-by-${hasApiSignal ? "api-signal" : "path-spread"}`));
|
|
159
241
|
return {
|
|
160
242
|
selectedRisk: "standard",
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { createHash, randomUUID } from "node:crypto";
|
|
2
|
-
import { mkdir, rename, rm, writeFile } from "node:fs/promises";
|
|
2
|
+
import { mkdir, readdir, rename, rm, readFile, writeFile } from "node:fs/promises";
|
|
3
3
|
import path from "node:path";
|
|
4
4
|
import { readFrontendToolProgress } from "./frontend-durable-tools.js";
|
|
5
5
|
import { frontendModelCapabilities } from "../../shared/frontend-execution-policy.js";
|
|
@@ -229,7 +229,7 @@ export async function observeFrontendSession(input, execute) {
|
|
|
229
229
|
await waitForWrite(save());
|
|
230
230
|
return result;
|
|
231
231
|
}
|
|
232
|
-
|
|
232
|
+
function uniqueFrontendSessions(receipts) {
|
|
233
233
|
const byId = new Map();
|
|
234
234
|
for (const receipt of receipts) {
|
|
235
235
|
const previous = byId.get(receipt.identity.sessionId);
|
|
@@ -240,10 +240,124 @@ export function summarizeFrontendSessions(receipts) {
|
|
|
240
240
|
if (!previous || difference > 0)
|
|
241
241
|
byId.set(receipt.identity.sessionId, receipt);
|
|
242
242
|
}
|
|
243
|
-
|
|
243
|
+
return [...byId.values()];
|
|
244
|
+
}
|
|
245
|
+
export function summarizeFrontendSessions(receipts) {
|
|
246
|
+
const unique = uniqueFrontendSessions(receipts);
|
|
244
247
|
return { sessions: unique.length, failedSessions: unique.filter(r => r.terminal?.ok === false).length, incompleteSessions: unique.filter(r => r.completion === "incomplete").length,
|
|
245
248
|
failureLayers: Object.fromEntries([...new Set(unique.flatMap(r => r.terminal && r.terminal.failureLayer !== "none" ? [r.terminal.failureLayer] : []))].sort().map(layer => [layer, unique.filter(r => r.terminal?.failureLayer === layer).length])),
|
|
246
249
|
progress: Object.fromEntries(["durableCommittedDelta", "durableSubmissions", "durableReplacements", "failedRecords", "duplicateRecords", "successfulReadBytes", "duplicateReadBytes"].map(key => [key, sumKnown(unique.map(r => r.progress[key]))])),
|
|
247
250
|
totalTokens: sumKnown(unique.flatMap(r => r.attempts.length ? r.attempts.map(a => a.actual?.totalTokens ?? null) : [r.actual.totalTokens])),
|
|
248
251
|
sessionDurationSumMs: sumKnown(unique.map(r => r.terminal?.durationMs ?? null)) };
|
|
249
252
|
}
|
|
253
|
+
/** Read the persisted session receipts for one Plan node. This is deliberately
|
|
254
|
+
* an observation-only projection: missing or malformed receipts remain
|
|
255
|
+
* missing, never zero, so a canary cannot manufacture an improvement. */
|
|
256
|
+
export async function readFrontendSessionBudgetReceipts(runDir, nodeId = "frontend-plan-pi") {
|
|
257
|
+
const root = path.join(runDir, nodeId);
|
|
258
|
+
const files = [];
|
|
259
|
+
const unreadable = [];
|
|
260
|
+
const visit = async (dir, insideBudget = false) => {
|
|
261
|
+
let entries;
|
|
262
|
+
try {
|
|
263
|
+
entries = await readdir(dir, { withFileTypes: true });
|
|
264
|
+
}
|
|
265
|
+
catch (error) {
|
|
266
|
+
if (error.code !== "ENOENT")
|
|
267
|
+
unreadable.push(dir);
|
|
268
|
+
return;
|
|
269
|
+
}
|
|
270
|
+
for (const entry of entries) {
|
|
271
|
+
const absolute = path.join(dir, entry.name);
|
|
272
|
+
if (entry.isDirectory() && (insideBudget || entry.name === "session-budget" || entry.name === "parallel" || /^coverage-\d+$/.test(entry.name)))
|
|
273
|
+
await visit(absolute, insideBudget || entry.name === "session-budget");
|
|
274
|
+
else if (insideBudget)
|
|
275
|
+
files.push(absolute);
|
|
276
|
+
}
|
|
277
|
+
};
|
|
278
|
+
await visit(root);
|
|
279
|
+
const receipts = [];
|
|
280
|
+
for (const file of [...files, ...unreadable].sort()) {
|
|
281
|
+
try {
|
|
282
|
+
const value = JSON.parse(await readFile(file, "utf8"));
|
|
283
|
+
if (!isCanaryReceipt(value))
|
|
284
|
+
throw new Error("Invalid session receipt");
|
|
285
|
+
receipts.push(value);
|
|
286
|
+
}
|
|
287
|
+
catch {
|
|
288
|
+
// Keep a visible incomplete receipt so canary aggregation cannot turn a
|
|
289
|
+
// truncated artifact into an apparent zero-cost improvement.
|
|
290
|
+
receipts.push({
|
|
291
|
+
schemaVersion: 1,
|
|
292
|
+
policyVersion: "frontend-session-observation-v1",
|
|
293
|
+
identity: { sessionId: `invalid:${file}`, phase: "unknown", scopeIds: [], taskId: null, runId: null, nodeId, attempt: null, sourceDigest: null, contractDigest: null },
|
|
294
|
+
startedAt: new Date(0).toISOString(), finishedAt: null, completion: "incomplete",
|
|
295
|
+
planned: { prompt: { sha256: "", chars: 0, bytes: 0, estimatedTokens: 0 }, userMessage: { sha256: "", chars: 0, bytes: 0, estimatedTokens: 0 }, toolSchemas: null, estimatorVersion: "unknown", capabilities: { ...frontendModelCapabilities(undefined), model: null }, unmeasuredComponents: [] },
|
|
296
|
+
actual: { ...collectFrontendUsage([]) }, attempts: [],
|
|
297
|
+
progress: { memoryCommittedDelta: null, durableCommittedDelta: null, failedRecords: null, duplicateRecords: null, durableSubmissions: null, durableReplacements: null, successfulReadBytes: null, duplicateReadBytes: null }, terminal: null,
|
|
298
|
+
});
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
return receipts;
|
|
302
|
+
}
|
|
303
|
+
export function summarizeFrontendPlanCanary(receipts) {
|
|
304
|
+
const unique = uniqueFrontendSessions(receipts);
|
|
305
|
+
const complete = unique.length > 0 && unique.every(receipt => isCanaryReceipt(receipt) && receipt.completion === "complete");
|
|
306
|
+
const known = (values, sum) => {
|
|
307
|
+
if (!complete || values.length === 0)
|
|
308
|
+
return null;
|
|
309
|
+
const present = values.filter((value) => value !== null && value !== undefined);
|
|
310
|
+
return present.length === values.length ? sum(present) : null;
|
|
311
|
+
};
|
|
312
|
+
return {
|
|
313
|
+
planModelRequests: known(unique.map((receipt) => receipt.actual.requestCount), (items) => items.reduce((total, value) => total + value, 0)),
|
|
314
|
+
planSessions: complete ? unique.length : null,
|
|
315
|
+
// Receipt envelope includes gaps and overlapping sessions only once. The
|
|
316
|
+
// run reader below replaces this with the authoritative node interval.
|
|
317
|
+
planWallClockMs: complete ? Math.max(...unique.map(r => Date.parse(r.finishedAt))) - Math.min(...unique.map(r => Date.parse(r.startedAt))) : null,
|
|
318
|
+
planTokens: known(unique.map((receipt) => receipt.actual.totalTokens), (items) => items.reduce((total, value) => total + value, 0)),
|
|
319
|
+
toolRejections: known(unique.map((receipt) => receipt.progress.failedRecords), (items) => items.reduce((total, value) => total + value, 0)),
|
|
320
|
+
retries: complete ? unique.reduce((total, receipt) => total + Math.max(0, receipt.attempts.length - 1), 0) : null,
|
|
321
|
+
};
|
|
322
|
+
}
|
|
323
|
+
/** Validate the fields consumed by aggregation before accessing nested values.
|
|
324
|
+
* Valid JSON with the wrong shape is missing evidence too. */
|
|
325
|
+
function isCanaryReceipt(value) {
|
|
326
|
+
const r = record(value), identity = record(r.identity), actual = record(r.actual), progress = record(r.progress);
|
|
327
|
+
const nullableCount = (v) => v === null || (typeof v === "number" && Number.isFinite(v) && Number.isInteger(v) && v >= 0);
|
|
328
|
+
return r.schemaVersion === 1 && r.policyVersion === "frontend-session-observation-v1"
|
|
329
|
+
&& typeof identity.sessionId === "string" && identity.sessionId.length > 0
|
|
330
|
+
&& typeof r.startedAt === "string" && Number.isFinite(Date.parse(r.startedAt))
|
|
331
|
+
&& (r.completion === "incomplete" || (r.completion === "complete" && typeof r.finishedAt === "string" && Date.parse(r.finishedAt) >= Date.parse(r.startedAt)))
|
|
332
|
+
&& Array.isArray(r.attempts) && r.attempts.every(a => a !== null && typeof a === "object")
|
|
333
|
+
&& nullableCount(actual.requestCount) && nullableCount(actual.totalTokens) && nullableCount(progress.failedRecords);
|
|
334
|
+
}
|
|
335
|
+
/** Read one frozen canary run. The optional artifact may add only run-level
|
|
336
|
+
* timing/review/recovery observations; Plan session counts, retries, tokens,
|
|
337
|
+
* request usage and tool rejections always come from session receipts. */
|
|
338
|
+
export async function readFrontendPlanCanaryRunMetrics(runDir, nodeId = "frontend-plan-pi") {
|
|
339
|
+
const sessionMetrics = summarizeFrontendPlanCanary(await readFrontendSessionBudgetReceipts(runDir, nodeId));
|
|
340
|
+
let supplemental = {};
|
|
341
|
+
try {
|
|
342
|
+
const parsed = JSON.parse(await readFile(path.join(runDir, "frontend-plan-canary-metrics.json"), "utf8"));
|
|
343
|
+
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
|
|
344
|
+
supplemental = parsed;
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
catch {
|
|
348
|
+
// Missing or malformed supplemental evidence remains unavailable.
|
|
349
|
+
}
|
|
350
|
+
const optionalNumber = (key) => {
|
|
351
|
+
const value = supplemental[key];
|
|
352
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 0
|
|
353
|
+
? value
|
|
354
|
+
: null;
|
|
355
|
+
};
|
|
356
|
+
return {
|
|
357
|
+
...sessionMetrics,
|
|
358
|
+
dagWallClockMs: optionalNumber("dagWallClockMs"),
|
|
359
|
+
designReviewRejectRate: optionalNumber("designReviewRejectRate"),
|
|
360
|
+
finalReviewDefectRate: optionalNumber("finalReviewDefectRate"),
|
|
361
|
+
recoveryCount: optionalNumber("recoveryCount"),
|
|
362
|
+
};
|
|
363
|
+
}
|