@tea-agent/loop-agent 0.44.0-next.8 → 0.44.0-next.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +97 -0
- package/dist/application/evaluation/budget.js +21 -0
- package/dist/application/evaluation/corpus-hash.js +10 -15
- package/dist/application/evaluation/corpus.js +2 -1
- package/dist/application/evaluation/frontend-browser-acceptance.js +77 -0
- package/dist/application/evaluation/frontend-corpus-browser-acceptance.js +175 -0
- package/dist/application/evaluation/frontend-gateway-evidence.js +226 -0
- package/dist/application/evaluation/frontend-pair-registration.js +143 -0
- package/dist/application/evaluation/frontend-paired-summary.js +150 -0
- package/dist/application/evaluation/frontend-run-observation.js +144 -0
- package/dist/application/evaluation/frontend-shared-evidence.js +105 -0
- package/dist/application/evaluation/types.js +45 -12
- package/dist/application/task-lifecycle/observe.js +26 -1
- package/dist/build-stamp.json +3 -3
- package/dist/cli/command-definitions.js +11 -0
- package/dist/cli/program.js +4 -0
- package/dist/commands/task-advance.js +3 -5
- package/dist/commands/task-source-prepare.js +13 -1
- package/dist/executors/dag-pi/tools/design-terminal-tools.js +17 -7
- package/dist/executors/dag-pi/tools/review-terminal-tools.js +17 -7
- package/dist/executors/dag-pi-executor.js +2048 -125
- package/dist/executors/pi-executor.js +6 -1
- package/dist/executors/pi-sdk-executor.js +76 -11
- package/dist/executors/shell-executor.js +88 -12
- package/dist/infrastructure/console/operation-store.js +2 -0
- package/dist/infrastructure/evaluation/corpus-store.js +37 -12
- package/dist/shared/operator/capabilities.js +10 -5
- package/dist/task/config-types.js +22 -9
- package/dist/task/contract/project.js +13 -0
- package/dist/task/contract/schema.js +2 -8
- package/dist/task/source-prepare/parse-intent.js +124 -5
- package/dist/task/source-prepare/prepare.js +6 -1
- package/dist/task/source-prepare/semantic-intake.js +17 -2
- package/dist/task/source-prepare/source-execution.js +65 -0
- package/dist/task/source-prepare/source-fidelity-pi.js +21 -8
- package/dist/task/source-prepare/source-provider-budget.js +290 -0
- package/dist/worker/console/operator-actions.js +0 -1
- package/dist/worker/console/operator-user-error.js +1 -1
- package/dist/worker/console/prd-intake-bridge.js +5 -15
- package/dist/workflows/dag/budget-enforcement.js +31 -2
- package/dist/workflows/dag/frontend-closeout.js +3 -1
- package/dist/workflows/dag/frontend-contract-facts.js +11 -8
- package/dist/workflows/dag/frontend-design-policy.js +6 -6
- package/dist/workflows/dag/frontend-durable-tools.js +90 -70
- package/dist/workflows/dag/frontend-implementation-contract.js +146 -59
- package/dist/workflows/dag/frontend-plan-canary.js +66 -16
- package/dist/workflows/dag/frontend-plan-decision-contract.js +13 -21
- package/dist/workflows/dag/frontend-plan-progress.js +92 -0
- package/dist/workflows/dag/frontend-recovery-plan.js +11 -10
- package/dist/workflows/dag/frontend-recovery-run.js +51 -46
- package/dist/workflows/dag/frontend-repair-assertions.js +36 -0
- package/dist/workflows/dag/frontend-review-context.js +29 -1
- package/dist/workflows/dag/frontend-review-scopes.js +92 -2
- package/dist/workflows/dag/frontend-risk.js +8 -3
- package/dist/workflows/dag/frontend-root-observation.js +261 -0
- package/dist/workflows/dag/frontend-session-budget.js +52 -39
- package/dist/workflows/dag/frontend-session-context.js +10 -0
- package/dist/workflows/dag/frontend-test-execution-evidence.js +206 -48
- package/dist/workflows/dag/frontend-typed-event-store.js +11 -6
- package/dist/workflows/dag/frontend-verification-trace.js +6 -0
- package/dist/workflows/dag/frontend-writer-admission.js +2 -1
- package/dist/workflows/dag/init-hybrid.js +108 -37
- package/dist/workflows/dag/node-execution.js +11 -11
- package/dist/workflows/dag/rerun-feedback.js +35 -4
- package/dist/workflows/dag/rerun-task.js +62 -67
- package/dist/workflows/dag/runner.js +58 -12
- package/dist/workflows/dag/types.js +15 -8
- package/docs/architecture/runtime-boundaries.md +1 -1
- package/docs/templates/backend-test-dag.json +4 -4
- package/docs/templates/frontend-implementation-contract.schema.json +31 -11
- package/package.json +1 -1
- package/skills/frontend-design-review/SKILL.md +8 -11
- package/skills/frontend-design-review/references/review-checklist.md +3 -4
- package/skills/frontend-review/SKILL.md +5 -1
- package/skills/loop-agent/references/command-reference.md +8 -0
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { parseFrontendTestExecutionReport } from "./frontend-test-execution-evidence.js";
|
|
2
|
+
import { readFileBoundarySafe } from "../../shared/path-safety.js";
|
|
1
3
|
import { createHash, randomUUID } from "node:crypto";
|
|
2
4
|
import { readFile, realpath, lstat } from "node:fs/promises";
|
|
3
5
|
import path from "node:path";
|
|
@@ -11,9 +13,10 @@ const isArtifactReference = (value) => isRecord(value)
|
|
|
11
13
|
&& typeof value.sha256 === "string" && /^[a-f0-9]{64}$/.test(value.sha256);
|
|
12
14
|
const hash = (text) => createHash("sha256").update(text).digest("hex");
|
|
13
15
|
/** The section index is navigation; reconstruct coverage against the canonical source before using it. */
|
|
14
|
-
export async function loadFrontendReviewScopes(runDir, phase, targetBytes) {
|
|
16
|
+
export async function loadFrontendReviewScopes(runDir, phase, targetBytes, workspaceRoot) {
|
|
15
17
|
const root = await realpath(runDir);
|
|
16
18
|
const files = new Map();
|
|
19
|
+
const sourceChecks = [];
|
|
17
20
|
async function read(relative, expected) {
|
|
18
21
|
const absolute = path.resolve(root, relative);
|
|
19
22
|
if (!absolute.startsWith(root + path.sep) || (await lstat(absolute)).isSymbolicLink() || !(await realpath(absolute)).startsWith(root + path.sep))
|
|
@@ -103,11 +106,88 @@ export async function loadFrontendReviewScopes(runDir, phase, targetBytes) {
|
|
|
103
106
|
if (diffIndex.patchSha256 !== payload.diff.patchSha256)
|
|
104
107
|
throw Error("FRONTEND_REVIEW_SCOPE_DRIFT: review diff binding");
|
|
105
108
|
scopes.push(context);
|
|
109
|
+
const reportCoverage = new Map();
|
|
110
|
+
const reportedTests = [];
|
|
111
|
+
for (const ref of payload.verificationTrace?.testExecutionEvidence ?? []) {
|
|
112
|
+
if (!isArtifactReference(ref) || typeof ref.reportPath !== "string" || !/^[a-f0-9]{64}$/.test(ref.reportSha256))
|
|
113
|
+
throw invalidScope("invalid test execution reference");
|
|
114
|
+
const excerpt = await read(ref.path, ref.sha256);
|
|
115
|
+
const data = JSON.parse(excerpt.text);
|
|
116
|
+
if (data.reportSha256 !== ref.reportSha256)
|
|
117
|
+
throw invalidScope("test execution report binding mismatch");
|
|
118
|
+
const coverage = reportCoverage.get(ref.reportPath) ?? { count: 0, total: data.total };
|
|
119
|
+
if (!Number.isInteger(data.total) || data.total < 1 || data.total !== coverage.total || data.offset !== coverage.count || !Array.isArray(data.tests) || !data.tests.length || data.tests.length > 24)
|
|
120
|
+
throw invalidScope("test report excerpt coverage invalid");
|
|
121
|
+
coverage.count += data.tests.length;
|
|
122
|
+
reportCoverage.set(ref.reportPath, coverage);
|
|
123
|
+
const report = await read(ref.reportPath, ref.reportSha256);
|
|
124
|
+
if (workspaceRoot) {
|
|
125
|
+
if (typeof data.cwd !== "string" || !Number.isInteger(data.offset) || data.offset < 0)
|
|
126
|
+
throw invalidScope("invalid test report excerpt coordinates");
|
|
127
|
+
const rootPath = await realpath(workspaceRoot);
|
|
128
|
+
const cwdPath = await realpath(path.resolve(workspaceRoot, data.cwd));
|
|
129
|
+
if (cwdPath !== rootPath && !cwdPath.startsWith(rootPath + path.sep))
|
|
130
|
+
throw invalidScope("test report cwd escaped workspace");
|
|
131
|
+
const actual = parseFrontendTestExecutionReport(JSON.parse(report.text), cwdPath, rootPath);
|
|
132
|
+
const expected = actual.slice(data.offset, data.offset + 24).map(test => ({ file: test.file, status: test.status, title: test.title.slice(0, 500), ...(test.title.length > 500 ? { titleTruncated: true } : {}) }));
|
|
133
|
+
if (data.total !== actual.length || !expected.length || sha256OfCanonicalJson(expected) !== sha256OfCanonicalJson(data.tests))
|
|
134
|
+
throw invalidScope("test report excerpt differs from actual runner report");
|
|
135
|
+
}
|
|
136
|
+
if (data.tests.some((test) => !isRecord(test) || typeof test.file !== "string" || typeof test.status !== "string"))
|
|
137
|
+
throw invalidScope("invalid test report records");
|
|
138
|
+
reportedTests.push(...data.tests);
|
|
139
|
+
scopes.push(excerpt);
|
|
140
|
+
}
|
|
141
|
+
if ([...reportCoverage.values()].some(coverage => coverage.count !== coverage.total))
|
|
142
|
+
throw invalidScope("test report excerpts incomplete");
|
|
143
|
+
const allowedTests = new Set((canonical.verificationTargets ?? []).filter((target) => target.mode === "behavior").map((target) => target.file));
|
|
144
|
+
// A trace may claim execution even when its entire report reference list
|
|
145
|
+
// was omitted. Require the actual, hash-checked reports for every behavior
|
|
146
|
+
// file before permitting review of the corresponding implementation.
|
|
147
|
+
for (const file of allowedTests) {
|
|
148
|
+
const normalized = path.posix.normalize(String(file).replace(/\\/g, "/"));
|
|
149
|
+
const matching = reportedTests.filter(test => path.posix.normalize(test.file) === normalized);
|
|
150
|
+
if (!workspaceRoot || !matching.length || matching.some(test => test.status !== "passed"))
|
|
151
|
+
throw invalidScope(`behavior test report coverage incomplete or not passed: ${String(file)}`);
|
|
152
|
+
}
|
|
153
|
+
if (allowedTests.size && (!Array.isArray(payload.testSources) || payload.testSources.length !== allowedTests.size || new Set(payload.testSources.map((source) => source.file)).size !== allowedTests.size))
|
|
154
|
+
throw invalidScope("contract test source coverage incomplete");
|
|
155
|
+
for (const source of payload.testSources ?? []) {
|
|
156
|
+
if (!allowedTests.has(source.file))
|
|
157
|
+
throw invalidScope("test source is outside frozen contract");
|
|
158
|
+
if (source.status === "unavailable")
|
|
159
|
+
continue;
|
|
160
|
+
if (source.status !== "captured" || !/^[a-f0-9]{64}$/.test(source.sha256) || !Array.isArray(source.scopes) || !source.scopes.length)
|
|
161
|
+
throw invalidScope("invalid test source reference");
|
|
162
|
+
let content = "";
|
|
163
|
+
for (const ref of source.scopes) {
|
|
164
|
+
if (!isArtifactReference(ref))
|
|
165
|
+
throw invalidScope("invalid test source slice");
|
|
166
|
+
const excerpt = await read(ref.path, ref.sha256);
|
|
167
|
+
const data = JSON.parse(excerpt.text);
|
|
168
|
+
if (data.file !== source.file || data.sourceSha256 !== source.sha256 || data.offset !== content.length || typeof data.text !== "string")
|
|
169
|
+
throw invalidScope("test source slice binding mismatch");
|
|
170
|
+
content += data.text;
|
|
171
|
+
scopes.push(excerpt);
|
|
172
|
+
}
|
|
173
|
+
if (hash(content) !== source.sha256)
|
|
174
|
+
throw invalidScope("test source snapshot incomplete");
|
|
175
|
+
if (workspaceRoot) {
|
|
176
|
+
const check = async () => {
|
|
177
|
+
const bytes = await readFileBoundarySafe({ boundaryRoot: workspaceRoot, targetPath: path.resolve(workspaceRoot, source.file), label: "frontend review test source" });
|
|
178
|
+
if (createHash("sha256").update(bytes).digest("hex") !== source.sha256)
|
|
179
|
+
throw Error(`FRONTEND_REVIEW_SCOPE_DRIFT: ${source.file}`);
|
|
180
|
+
};
|
|
181
|
+
await check();
|
|
182
|
+
sourceChecks.push(check);
|
|
183
|
+
}
|
|
184
|
+
}
|
|
106
185
|
for (const hunk of diffIndex.hunks)
|
|
107
186
|
scopes.push(await read(hunk.patchPath, hunk.sha256));
|
|
108
187
|
}
|
|
109
188
|
return { scopes, digest: sha256OfCanonicalJson([...files]), validate: async () => { for (const [relative, expected] of files)
|
|
110
|
-
await read(relative, expected);
|
|
189
|
+
await read(relative, expected); for (const check of sourceChecks)
|
|
190
|
+
await check(); } };
|
|
111
191
|
}
|
|
112
192
|
/** Checkpoints attest explicit model completion of already supplied, hash-bound scopes. */
|
|
113
193
|
export function createFrontendReviewScopeProtocol(input) {
|
|
@@ -118,6 +198,16 @@ export function createFrontendReviewScopeProtocol(input) {
|
|
|
118
198
|
return {
|
|
119
199
|
setActiveScope(ids) { active = new Set(ids); },
|
|
120
200
|
completedScopeIds,
|
|
201
|
+
findingScopeIds(previous) {
|
|
202
|
+
if (!input.inventory)
|
|
203
|
+
return undefined;
|
|
204
|
+
const ids = previous ?? [...active];
|
|
205
|
+
if (!ids.length || ids.some(id => !active.has(id) || !input.inventory.scopes.some(scope => scope.id === id)))
|
|
206
|
+
throw Object.assign(Error("Findings require their original active review scope"), { code: "FRONTEND_INPUT_SCOPE_VIOLATION" });
|
|
207
|
+
if (ids.some(id => completedScopeIds().has(id)))
|
|
208
|
+
throw Object.assign(Error("A completed review scope cannot be changed; restart its owning review"), { code: "FRONTEND_REVIEW_SCOPE_COMPLETED" });
|
|
209
|
+
return [...ids];
|
|
210
|
+
},
|
|
121
211
|
committedFacts: () => readCommittedEvents(input.getStore(), input.attemptId),
|
|
122
212
|
assertComplete() { const missing = input.inventory?.scopes.filter(s => !completedScopeIds().has(s.id)) ?? []; if (missing.length)
|
|
123
213
|
throw Object.assign(Error(`FRONTEND_REVIEW_SCOPE_INCOMPLETE: ${missing.map(s => s.id).join(", ")}`), { code: "FRONTEND_REVIEW_SCOPE_INCOMPLETE" }); },
|
|
@@ -55,7 +55,12 @@ function isExplicitSharedExclusion(clause, pattern) {
|
|
|
55
55
|
const prefix = clause.slice(0, match.index).trim();
|
|
56
56
|
const suffix = clause.slice(match.index + match[0].length).trim();
|
|
57
57
|
const actionInSignal = ["new-route", "new-dependency"].includes(pattern.id);
|
|
58
|
-
|
|
58
|
+
// A coordinated toolchain object remains under the same explicit negation.
|
|
59
|
+
// Match the entire suffix: qualifications ("unless..." / "除非...") and
|
|
60
|
+
// consequences must still retain the signal.
|
|
61
|
+
const excludedToolchainObject = pattern.id === "new-dependency" &&
|
|
62
|
+
/^(?:或|和|及|与)\s*(?:构建|打包)(?:系统|工具|流程)$/.test(suffix);
|
|
63
|
+
if (suffix && !excludedToolchainObject && !(actionInSignal && /^no$/i.test(prefix) && /^is\s+(?:required|needed)$/i.test(suffix)))
|
|
59
64
|
return false;
|
|
60
65
|
if (/^(?:不涉及|不(?:修改|变更|调整|改变|新增|添加)|无需(?:修改|变更|调整|改变|新增|添加)|no\s+changes?\s+to|(?:do\s+not|don't)\s+(?:change|modify))$/i.test(prefix))
|
|
61
66
|
return true;
|
|
@@ -169,13 +174,13 @@ export function classifyFrontendRisk(input) {
|
|
|
169
174
|
profile.includes("supervised") ||
|
|
170
175
|
/\bsupervised\b/i.test(blob);
|
|
171
176
|
for (const pattern of HIGH_RISK_PATTERNS) {
|
|
172
|
-
if (
|
|
177
|
+
if (hasAffirmativeSharedSignal(blob, pattern))
|
|
173
178
|
signals.push(pattern.id);
|
|
174
179
|
}
|
|
175
180
|
const paths = input.allowedPaths ?? [];
|
|
176
181
|
const spread = pathSpreadScore(paths);
|
|
177
182
|
if (paths.some((p) => /package\.json|pnpm-lock|yarn\.lock|package-lock/.test(p))) {
|
|
178
|
-
if (
|
|
183
|
+
if (hasAffirmativeSharedSignal(blob, { id: "new-dependency", re: DEPENDENCY_INSTALL_LANGUAGE })) {
|
|
179
184
|
signals.push("new-dependency");
|
|
180
185
|
}
|
|
181
186
|
else {
|
|
@@ -0,0 +1,261 @@
|
|
|
1
|
+
import { sourceBudgetContractIdentity } from "../../application/evaluation/budget.js";
|
|
2
|
+
import { verifyFrozenSourceUsage } from "../../task/source-prepare/source-provider-budget.js";
|
|
3
|
+
import { sha256OfCanonicalJson } from "../../task/contract/hash.js";
|
|
4
|
+
import { locateDagRun, readDagRunSpec, readDagRunState } from "./lifecycle.js";
|
|
5
|
+
import { validateCumulativeBudgetSuccessor } from "./budget-enforcement.js";
|
|
6
|
+
import { readFrontendSessionBudgetReceipts, summarizeFrontendPlanCanary, summarizeFrontendSessions } from "./frontend-session-budget.js";
|
|
7
|
+
/** One explicit recovery chain. Missing links never look like cheaper runs. */
|
|
8
|
+
export async function readFrontendRootObservation(workspaceRoot, runId) {
|
|
9
|
+
const diagnostics = [];
|
|
10
|
+
const chain = [];
|
|
11
|
+
const seen = new Set();
|
|
12
|
+
let sourceUsage = null;
|
|
13
|
+
let sourceBudgetHash;
|
|
14
|
+
const load = async (id) => {
|
|
15
|
+
if (!/^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(id))
|
|
16
|
+
throw Error("invalid run identity");
|
|
17
|
+
const found = await locateDagRun(workspaceRoot, id);
|
|
18
|
+
if (!found)
|
|
19
|
+
throw Error(`missing recovery run: ${id}`);
|
|
20
|
+
const state = await readDagRunState(found.runDir);
|
|
21
|
+
if (state.runId !== id || state.executionMode !== "execute")
|
|
22
|
+
throw Error(`invalid execution identity: ${id}`);
|
|
23
|
+
return { runDir: found.runDir, state };
|
|
24
|
+
};
|
|
25
|
+
const parentOf = (state) => {
|
|
26
|
+
const continuation = state.continuation?.parentRunId;
|
|
27
|
+
const recovery = state.frontendRecoveryState?.parentRunId;
|
|
28
|
+
if (continuation === state.runId)
|
|
29
|
+
throw Error("continuation ancestor cycle");
|
|
30
|
+
const recoveryParent = recovery === state.runId ? undefined : recovery;
|
|
31
|
+
if (continuation && recoveryParent && continuation !== recoveryParent)
|
|
32
|
+
throw Error("conflicting recovery parents");
|
|
33
|
+
return continuation ?? recoveryParent;
|
|
34
|
+
};
|
|
35
|
+
try {
|
|
36
|
+
let current = await load(runId);
|
|
37
|
+
const ancestorChildren = new Map();
|
|
38
|
+
// In a root's pending recovery intent parentRunId can refer to itself.
|
|
39
|
+
while (true) {
|
|
40
|
+
if (seen.has(current.state.runId))
|
|
41
|
+
throw Error("recovery ancestor cycle");
|
|
42
|
+
seen.add(current.state.runId);
|
|
43
|
+
const parent = parentOf(current.state);
|
|
44
|
+
if (!parent)
|
|
45
|
+
break;
|
|
46
|
+
ancestorChildren.set(parent, current.state.runId);
|
|
47
|
+
current = await load(parent);
|
|
48
|
+
}
|
|
49
|
+
const rootId = current.state.runId;
|
|
50
|
+
seen.clear();
|
|
51
|
+
let sourceHash;
|
|
52
|
+
while (true) {
|
|
53
|
+
if (seen.has(current.state.runId))
|
|
54
|
+
throw Error("recovery descendant cycle");
|
|
55
|
+
seen.add(current.state.runId);
|
|
56
|
+
chain.push(current);
|
|
57
|
+
const spec = await readDagRunSpec(current.runDir);
|
|
58
|
+
if (!spec.sourceBinding)
|
|
59
|
+
throw Error("missing frozen source binding");
|
|
60
|
+
const hash = sha256OfCanonicalJson(spec.sourceBinding);
|
|
61
|
+
if (sourceHash && hash !== sourceHash)
|
|
62
|
+
throw Error("recovery source binding drift");
|
|
63
|
+
sourceHash = hash;
|
|
64
|
+
const boundBudgetHash = sha256OfCanonicalJson(spec.sourceProviderBudget ?? null);
|
|
65
|
+
if (sourceBudgetHash !== undefined && sourceBudgetHash !== boundBudgetHash)
|
|
66
|
+
throw Error("recovery source budget evidence drift");
|
|
67
|
+
sourceBudgetHash = boundBudgetHash;
|
|
68
|
+
if (spec.sourceProviderBudget) {
|
|
69
|
+
if (sha256OfCanonicalJson(spec.sourceProviderBudget.contract ?? null) !== sha256OfCanonicalJson(sourceBudgetContractIdentity(spec.taskContractBinding) ?? null) || spec.sourceProviderBudget.sourceBindingSha256 !== hash || sha256OfCanonicalJson(current.state.sourceProviderBudget ?? null) !== boundBudgetHash)
|
|
70
|
+
throw Error("source budget state/spec binding mismatch");
|
|
71
|
+
sourceUsage = await verifyFrozenSourceUsage(workspaceRoot, spec.sourceProviderBudget);
|
|
72
|
+
diagnostics.push(...sourceUsage.diagnostics);
|
|
73
|
+
if (sourceUsage.complete && ((current.state.budgetLedger?.consumed.tokens ?? -1) < sourceUsage.totalTokens || (current.state.budgetLedger?.consumed.providerRequests ?? -1) < spec.sourceProviderBudget.ledger.reservedRequests))
|
|
74
|
+
throw Error("source cost missing from cumulative budget");
|
|
75
|
+
}
|
|
76
|
+
const recovery = current.state.frontendRecoveryState;
|
|
77
|
+
if (recovery && recovery.recoveryRootRunId !== rootId)
|
|
78
|
+
throw Error("recovery root identity drift");
|
|
79
|
+
if (!["finished", "failed", "partial_failed"].includes(current.state.status) || !current.state.finishedAt)
|
|
80
|
+
throw Error(`incomplete run: ${current.state.runId}`);
|
|
81
|
+
const declaredChild = recovery?.childRunId === current.state.runId ? undefined : recovery?.childRunId;
|
|
82
|
+
const selectedChild = ancestorChildren.get(current.state.runId);
|
|
83
|
+
if (declaredChild && selectedChild && declaredChild !== selectedChild)
|
|
84
|
+
throw Error("conflicting recovery children");
|
|
85
|
+
const child = selectedChild ?? declaredChild;
|
|
86
|
+
if (!child)
|
|
87
|
+
break;
|
|
88
|
+
const next = await load(child);
|
|
89
|
+
if (parentOf(next.state) !== current.state.runId)
|
|
90
|
+
throw Error("recovery parent/child mismatch");
|
|
91
|
+
validateCumulativeBudgetSuccessor(current.state, next.state);
|
|
92
|
+
current = next;
|
|
93
|
+
}
|
|
94
|
+
if (!seen.has(runId))
|
|
95
|
+
throw Error("requested run is outside the root chain");
|
|
96
|
+
}
|
|
97
|
+
catch (error) {
|
|
98
|
+
diagnostics.push(error instanceof Error ? error.message : String(error));
|
|
99
|
+
}
|
|
100
|
+
const receipts = [];
|
|
101
|
+
const identities = new Map();
|
|
102
|
+
for (const item of chain) {
|
|
103
|
+
const own = await readFrontendSessionBudgetReceipts(item.runDir);
|
|
104
|
+
const executedPlan = item.state.nodes?.["frontend-plan-pi"];
|
|
105
|
+
if (!own.length && executedPlan && executedPlan.origin?.kind !== "imported")
|
|
106
|
+
diagnostics.push(`missing Plan receipts: ${item.state.runId}`);
|
|
107
|
+
for (const receipt of own) {
|
|
108
|
+
if (!chain.some(entry => entry.state.runId === receipt.identity.runId))
|
|
109
|
+
diagnostics.push("receipt outside recovery lineage");
|
|
110
|
+
const key = `${receipt.identity.runId}:${receipt.identity.sessionId}`;
|
|
111
|
+
const digest = sha256OfCanonicalJson(receipt);
|
|
112
|
+
if (identities.has(key) && identities.get(key) !== digest)
|
|
113
|
+
diagnostics.push(`conflicting receipt: ${key}`);
|
|
114
|
+
identities.set(key, digest);
|
|
115
|
+
receipts.push(receipt);
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
if (receipts.some(receipt => receipt.completion !== "complete"))
|
|
119
|
+
diagnostics.push("incomplete session receipts");
|
|
120
|
+
let complete = diagnostics.length === 0 && chain.length > 0;
|
|
121
|
+
const plan = summarizeFrontendPlanCanary(complete ? receipts : []);
|
|
122
|
+
const first = chain[0]?.state, last = chain.at(-1)?.state;
|
|
123
|
+
const wall = first && last ? Date.parse(last.finishedAt) - Date.parse(first.startedAt) : NaN;
|
|
124
|
+
if (complete && (!Number.isFinite(wall) || wall < 0))
|
|
125
|
+
diagnostics.push("invalid root timing");
|
|
126
|
+
complete = diagnostics.length === 0 && complete;
|
|
127
|
+
return { complete, diagnostics, sourceUsage, runIds: chain.map(entry => entry.state.runId), plan,
|
|
128
|
+
sessions: summarizeFrontendSessions(receipts),
|
|
129
|
+
recoveryCount: complete ? chain.length - 1 : null,
|
|
130
|
+
dagWallClockMs: complete && Number.isFinite(wall) && wall >= 0 ? wall : null,
|
|
131
|
+
// The final ledger already includes inherited consumption. Never sum it.
|
|
132
|
+
cumulativeBudget: complete ? last?.budgetLedger ?? null : null,
|
|
133
|
+
outcome: complete ? last?.status ?? null : null };
|
|
134
|
+
}
|
|
135
|
+
/** Explicit identities are required; absent on both sides does not mean equal. */
|
|
136
|
+
export function compareFrontendPairIdentity(before, after) {
|
|
137
|
+
return ["protocol", "source", "model", "budget", "initialTree", "controller", "promptTools", "acceptance"].flatMap(key => typeof before[key] !== "string" || !before[key]?.trim() || typeof after[key] !== "string" || !after[key]?.trim() ? [`missing-pair-identity:${key}`] : before[key] !== after[key] ? [`pair-identity-mismatch:${key}`] : []);
|
|
138
|
+
}
|
|
139
|
+
export const FRONTEND_PLAN_CANARY_SCENARIOS = [
|
|
140
|
+
"missing-idle-loading-empty",
|
|
141
|
+
"loading-not-mounted",
|
|
142
|
+
"test-layer-does-not-prove-wiring",
|
|
143
|
+
"mock-claimed-as-real-integration",
|
|
144
|
+
];
|
|
145
|
+
export async function readFrontendCanaryEvidence(input) {
|
|
146
|
+
const { createHash } = await import("node:crypto");
|
|
147
|
+
const path = await import("node:path");
|
|
148
|
+
const { z } = await import("zod");
|
|
149
|
+
const { readFileBoundarySafe } = await import("../../shared/path-safety.js");
|
|
150
|
+
const { typedEventRecordSchema } = await import("./frontend-typed-event-store.js");
|
|
151
|
+
const ref = z.object({ path: z.string().min(1), sha256: z.string().regex(/^[a-f0-9]{64}$/) }).strict();
|
|
152
|
+
const identitySchema = z.object({ protocol: ref, source: ref, model: ref, budget: ref, initialTree: ref, controller: ref, promptTools: ref, acceptance: ref }).strict();
|
|
153
|
+
const manifestSchema = z.object({ schemaVersion: z.literal(1), runId: z.string().min(1), run: ref, state: ref, identity: identitySchema, acceptance: ref }).strict();
|
|
154
|
+
const read = async (value, boundaryRoot = input.manifest.directory) => {
|
|
155
|
+
const checked = ref.parse({ path: value.path, sha256: value.sha256 });
|
|
156
|
+
if (path.isAbsolute(checked.path) || checked.path.split(/[\\/]/).includes(".."))
|
|
157
|
+
throw Error("canary-evidence-path");
|
|
158
|
+
const bytes = await readFileBoundarySafe({ boundaryRoot, targetPath: path.join(boundaryRoot, checked.path), label: "canary evidence" });
|
|
159
|
+
if (createHash("sha256").update(bytes).digest("hex") !== checked.sha256)
|
|
160
|
+
throw Error(`canary-evidence-hash:${checked.path}`);
|
|
161
|
+
return bytes;
|
|
162
|
+
};
|
|
163
|
+
const json = async (value) => JSON.parse((await read(value)).toString("utf8"));
|
|
164
|
+
const manifest = manifestSchema.parse(await json(input.manifest));
|
|
165
|
+
const tailId = input.runIds.at(-1);
|
|
166
|
+
const tail = tailId && await locateDagRun(input.workspaceRoot, tailId);
|
|
167
|
+
if (!tail || manifest.runId !== tailId || path.resolve(tail.runDir) !== path.resolve(input.runDir))
|
|
168
|
+
throw Error("canary-evidence-run-mismatch");
|
|
169
|
+
if (manifest.run.path !== "run.json" || manifest.state.path !== "state.json")
|
|
170
|
+
throw Error("canary-evidence-run-path");
|
|
171
|
+
await read(manifest.run, input.runDir);
|
|
172
|
+
await read(manifest.state, input.runDir);
|
|
173
|
+
const spec = await readDagRunSpec(input.runDir);
|
|
174
|
+
const state = await readDagRunState(input.runDir);
|
|
175
|
+
if (state.runId !== tailId || state.status !== "finished" || !state.finishedAt)
|
|
176
|
+
throw Error("canary-evidence-run-not-finished");
|
|
177
|
+
const identity = {};
|
|
178
|
+
for (const key of Object.keys(manifest.identity)) {
|
|
179
|
+
await read(manifest.identity[key]);
|
|
180
|
+
identity[key] = manifest.identity[key].sha256;
|
|
181
|
+
}
|
|
182
|
+
if (sha256OfCanonicalJson(await json(manifest.identity.source)) !== sha256OfCanonicalJson(spec.sourceBinding))
|
|
183
|
+
throw Error("canary-evidence-source-mismatch");
|
|
184
|
+
if (sha256OfCanonicalJson(await json(manifest.identity.budget)) !== sha256OfCanonicalJson(spec.budget ?? null))
|
|
185
|
+
throw Error("canary-evidence-budget-mismatch");
|
|
186
|
+
const modelConfig = z.object({ models: z.array(z.string().min(1)).min(1), parameters: z.record(z.unknown()).optional() }).strict().parse(await json(manifest.identity.model));
|
|
187
|
+
const outcome = z.enum(["pass", "fail"]);
|
|
188
|
+
const expected = z.object({ cases: z.array(z.object({ id: z.string().min(1), expected: outcome, scenario: z.enum(FRONTEND_PLAN_CANARY_SCENARIOS).optional() }).strict()).min(1) }).strict().parse(await json(manifest.identity.acceptance));
|
|
189
|
+
const actual = z.object({ schemaVersion: z.literal(1), runId: z.string(), run: ref, state: ref, results: z.array(z.object({ id: z.string().min(1), actual: outcome, evidence: ref, scenarioResult: z.enum(["blocked", "reported", "not-observed"]).optional() }).strict()).min(1) }).strict().parse(await json(manifest.acceptance));
|
|
190
|
+
if (actual.runId !== tailId || sha256OfCanonicalJson(actual.run) !== sha256OfCanonicalJson(manifest.run) || sha256OfCanonicalJson(actual.state) !== sha256OfCanonicalJson(manifest.state))
|
|
191
|
+
throw Error("canary-acceptance-run-mismatch");
|
|
192
|
+
const expectedIds = new Set(expected.cases.map(row => row.id));
|
|
193
|
+
const actualIds = new Set(actual.results.map(row => row.id));
|
|
194
|
+
if (expectedIds.size !== expected.cases.length || actualIds.size !== actual.results.length || expectedIds.size !== actualIds.size || [...actualIds].some(id => !expectedIds.has(id)))
|
|
195
|
+
throw Error("canary-acceptance-case-mismatch");
|
|
196
|
+
for (const result of actual.results)
|
|
197
|
+
await read(result.evidence);
|
|
198
|
+
const defects = expected.cases.filter(row => actual.results.find(result => result.id === row.id).actual !== row.expected).length;
|
|
199
|
+
const regressions = {};
|
|
200
|
+
for (const scenario of FRONTEND_PLAN_CANARY_SCENARIOS) {
|
|
201
|
+
const cases = expected.cases.filter(row => row.scenario === scenario);
|
|
202
|
+
const verified = cases.length > 0 && cases.every(row => {
|
|
203
|
+
const result = actual.results.find(value => value.id === row.id);
|
|
204
|
+
return result.actual === row.expected && (result.scenarioResult === "blocked" || result.scenarioResult === "reported");
|
|
205
|
+
});
|
|
206
|
+
regressions[scenario] = verified ? "reported" : "not-observed";
|
|
207
|
+
}
|
|
208
|
+
let reviews = 0, rejects = 0;
|
|
209
|
+
const actualModels = new Set();
|
|
210
|
+
for (const id of input.runIds) {
|
|
211
|
+
const found = await locateDagRun(input.workspaceRoot, id);
|
|
212
|
+
if (!found)
|
|
213
|
+
throw Error("canary-design-run-missing");
|
|
214
|
+
for (const receipt of await readFrontendSessionBudgetReceipts(found.runDir)) {
|
|
215
|
+
if (receipt.identity.executionKind !== "model")
|
|
216
|
+
continue;
|
|
217
|
+
const plannedModel = receipt.planned?.capabilities?.model;
|
|
218
|
+
const models = [plannedModel, ...receipt.attempts.map(attempt => attempt.model)];
|
|
219
|
+
if (!receipt.attempts.length || models.some(model => !model || !modelConfig.models.includes(model)))
|
|
220
|
+
throw Error("canary-evidence-model-mismatch");
|
|
221
|
+
for (const attempt of receipt.attempts)
|
|
222
|
+
actualModels.add(attempt.model);
|
|
223
|
+
}
|
|
224
|
+
const runState = await readDagRunState(found.runDir);
|
|
225
|
+
const runSpec = await readDagRunSpec(found.runDir);
|
|
226
|
+
const node = runState.nodes["frontend-design-review-pi"];
|
|
227
|
+
if (!node && runSpec.tasks.some(task => task.id === "frontend-design-review-pi"))
|
|
228
|
+
throw Error("canary-design-node-missing");
|
|
229
|
+
// Small/micro paths have no independent design review; skipped/imported
|
|
230
|
+
// nodes contribute no new verdict. Missing facts on executed nodes are errors.
|
|
231
|
+
if (!node || node.status === "SKIPPED" || node.origin?.kind === "imported")
|
|
232
|
+
continue;
|
|
233
|
+
const bytes = await readFileBoundarySafe({ boundaryRoot: found.runDir, targetPath: path.join(found.runDir, "frontend-design-review-pi/design-typed-facts.jsonl"), label: "canary design facts" });
|
|
234
|
+
const records = bytes.toString("utf8").split("\n").filter(line => line.trim()).map(line => typedEventRecordSchema.parse(JSON.parse(line)));
|
|
235
|
+
const terminals = records.filter(record => record.phase === "committed" && ["approve_design", "request_design_changes"].includes(String(record.fact.kind)));
|
|
236
|
+
const attempts = new Set();
|
|
237
|
+
if (!terminals.length)
|
|
238
|
+
throw Error("canary-design-terminal-missing");
|
|
239
|
+
for (const record of terminals) {
|
|
240
|
+
if (record.payloadSha256 !== sha256OfCanonicalJson(record.fact) || attempts.has(record.attemptId))
|
|
241
|
+
throw Error("canary-design-terminal-invalid");
|
|
242
|
+
attempts.add(record.attemptId);
|
|
243
|
+
reviews++;
|
|
244
|
+
if (record.fact.kind === "request_design_changes")
|
|
245
|
+
rejects++;
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
if (actualModels.size !== 1)
|
|
249
|
+
throw Error("canary-evidence-mixed-models");
|
|
250
|
+
identity.model = sha256OfCanonicalJson({ configurationSha256: identity.model, actualModel: [...actualModels][0] });
|
|
251
|
+
// Recheck all observed files after collection, including run state/spec.
|
|
252
|
+
await read(input.manifest);
|
|
253
|
+
await read(manifest.run, input.runDir);
|
|
254
|
+
await read(manifest.state, input.runDir);
|
|
255
|
+
for (const value of Object.values(manifest.identity))
|
|
256
|
+
await read(value);
|
|
257
|
+
await read(manifest.acceptance);
|
|
258
|
+
for (const result of actual.results)
|
|
259
|
+
await read(result.evidence);
|
|
260
|
+
return { identity, regressions, designReviewRejectRate: reviews ? rejects / reviews : 0, finalReviewDefectRate: defects / expected.cases.length, acceptancePassed: defects === 0 };
|
|
261
|
+
}
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { currentSourceExecution } from "../../task/source-prepare/source-execution.js";
|
|
2
|
+
import { withFrontendSessionIdentity } from "./frontend-session-context.js";
|
|
1
3
|
import { createHash, randomUUID } from "node:crypto";
|
|
2
4
|
import { mkdir, readdir, rename, rm, readFile, writeFile } from "node:fs/promises";
|
|
3
5
|
import path from "node:path";
|
|
@@ -157,13 +159,13 @@ export async function observeFrontendSession(input, execute) {
|
|
|
157
159
|
const before = safeObservation(() => input.committedCount?.(), undefined);
|
|
158
160
|
const toolsBefore = readFrontendToolProgress(input.customTools);
|
|
159
161
|
const receipt = {
|
|
160
|
-
schemaVersion:
|
|
161
|
-
identity: { sessionId: randomUUID(), phase: input.phase, scopeIds: input.scopeIds ?? [], taskId: input.taskId ?? null, runId: input.runId ?? null, nodeId: input.nodeId ?? null, attempt: input.attempt ?? null, sourceDigest: input.sourceDigest ?? null, contractDigest: input.contractDigest ?? null },
|
|
162
|
+
schemaVersion: 2, policyVersion: "frontend-session-observation-v2",
|
|
163
|
+
identity: { ...(currentSourceExecution(input.taskId) ? { sourceExecutionId: currentSourceExecution(input.taskId) } : {}), sessionId: randomUUID(), phase: input.phase, executionKind: input.executionKind ?? "model", dispatchReason: input.dispatchReason ?? "initial", scopeIds: input.scopeIds ?? [], taskId: input.taskId ?? null, runId: input.runId ?? null, nodeId: input.nodeId ?? null, attempt: input.attempt ?? null, sourceDigest: input.sourceDigest ?? null, contractDigest: input.contractDigest ?? null },
|
|
162
164
|
startedAt: new Date().toISOString(), finishedAt: null, completion: "incomplete",
|
|
163
165
|
planned: { prompt: measure(input.prompt), userMessage: measure(input.userMessage ?? ""), toolSchemas: input.customTools ? safeObservation(() => measure(JSON.stringify(input.customTools.map(t => { const tool = record(t); return { name: tool.name, description: tool.description, parameters: tool.parameters }; }))), null) : null,
|
|
164
166
|
estimatorVersion: "utf8-bytes-div3-v1-estimate", capabilities: { ...frontendModelCapabilities(undefined), model: input.model ?? null }, unmeasuredComponents: ["sdk-system-overhead", "built-in-tools", "history", "tool-results", "attachments"] },
|
|
165
167
|
actual: collectFrontendUsage([]), attempts: [],
|
|
166
|
-
progress: { memoryCommittedDelta: null, durableCommittedDelta: null, failedRecords: null, duplicateRecords: null, durableSubmissions: null, durableReplacements: null, successfulReadBytes: null, duplicateReadBytes: null }, terminal: null,
|
|
168
|
+
progress: { finalizeRejections: null, memoryCommittedDelta: null, durableCommittedDelta: null, failedRecords: null, duplicateRecords: null, durableSubmissions: null, durableReplacements: null, successfulReadBytes: null, duplicateReadBytes: null }, terminal: null,
|
|
167
169
|
};
|
|
168
170
|
let writes = Promise.resolve();
|
|
169
171
|
let abandoned = false;
|
|
@@ -203,14 +205,14 @@ export async function observeFrontendSession(input, execute) {
|
|
|
203
205
|
}
|
|
204
206
|
};
|
|
205
207
|
await waitForWrite(save());
|
|
206
|
-
const result = await execute(event => {
|
|
208
|
+
const result = await withFrontendSessionIdentity({ ...receipt.identity, scopeIds: [...receipt.identity.scopeIds] }, () => execute(event => {
|
|
207
209
|
const index = receipt.attempts.findIndex(a => a.id === event.id);
|
|
208
210
|
if (index < 0)
|
|
209
211
|
receipt.attempts.push(event);
|
|
210
212
|
else
|
|
211
213
|
receipt.attempts[index] = event;
|
|
212
214
|
void save();
|
|
213
|
-
});
|
|
215
|
+
}));
|
|
214
216
|
receipt.finishedAt = new Date().toISOString();
|
|
215
217
|
receipt.completion = "complete";
|
|
216
218
|
receipt.actual = receipt.attempts.length ? aggregateAttempts(receipt.attempts) : result.usageObservation ?? collectFrontendUsage([]);
|
|
@@ -220,8 +222,10 @@ export async function observeFrontendSession(input, execute) {
|
|
|
220
222
|
receipt.progress.durableCommittedDelta = durableBefore !== undefined && durableAfter !== undefined ? Math.max(0, durableAfter - durableBefore) : null;
|
|
221
223
|
const toolsAfter = readFrontendToolProgress(input.customTools);
|
|
222
224
|
if (toolsBefore && toolsAfter)
|
|
223
|
-
for (const key of ["failedRecords", "duplicateRecords", "durableSubmissions", "durableReplacements"])
|
|
225
|
+
for (const key of ["finalizeRejections", "failedRecords", "duplicateRecords", "durableSubmissions", "durableReplacements"])
|
|
224
226
|
receipt.progress[key] = Math.max(0, toolsAfter[key] - toolsBefore[key]);
|
|
227
|
+
if (input.executionKind === "runtime")
|
|
228
|
+
receipt.progress.finalizeRejections = result.ok ? 0 : 1;
|
|
225
229
|
const reads = receipt.attempts.length ? receipt.attempts.map(a => a.reads) : [result.readObservation];
|
|
226
230
|
for (const key of ["successfulReadBytes", "duplicateReadBytes"])
|
|
227
231
|
receipt.progress[key] = sumKnown(reads.map(r => r?.[key] ?? null));
|
|
@@ -229,16 +233,30 @@ export async function observeFrontendSession(input, execute) {
|
|
|
229
233
|
await waitForWrite(save());
|
|
230
234
|
return result;
|
|
231
235
|
}
|
|
236
|
+
/** Observe deterministic compilation without inflating model/session counts. */
|
|
237
|
+
export async function observeFrontendFinalization(input, execute) {
|
|
238
|
+
let value;
|
|
239
|
+
await observeFrontendSession({ ...input, executionKind: "runtime", phase: "plan/runtime-finalize", prompt: "" }, async () => {
|
|
240
|
+
value = await execute();
|
|
241
|
+
const details = record(record(value).details ?? value);
|
|
242
|
+
return { ok: details.ok === true, assistantText: "", command: [], durationMs: 0, exitCode: null,
|
|
243
|
+
failureCategory: details.ok === true ? "success" : "invalid-output", modelDisplay: "runtime", parsedEvents: 0,
|
|
244
|
+
stderr: details.ok === true ? "" : String(details.error ?? details.code ?? "finalize rejected"), stdout: "", timedOut: false,
|
|
245
|
+
attemptedModels: [], fallbackUsed: false, tokensUsed: 0 };
|
|
246
|
+
});
|
|
247
|
+
return value;
|
|
248
|
+
}
|
|
232
249
|
function uniqueFrontendSessions(receipts) {
|
|
233
250
|
const byId = new Map();
|
|
234
251
|
for (const receipt of receipts) {
|
|
235
|
-
const
|
|
252
|
+
const key = `${receipt.identity.runId ?? "unknown"}:${receipt.identity.sessionId}`;
|
|
253
|
+
const previous = byId.get(key);
|
|
236
254
|
const rank = (r) => [r.completion === "complete" ? 1 : 0, Date.parse(r.finishedAt ?? r.startedAt), r.attempts.filter(a => a.completion === "complete").length, r.attempts.length];
|
|
237
255
|
const candidateRank = rank(receipt);
|
|
238
256
|
const previousRank = previous ? rank(previous) : [];
|
|
239
257
|
const difference = candidateRank.map((v, i) => v - (previousRank[i] ?? 0)).find(v => v !== 0) ?? 0;
|
|
240
258
|
if (!previous || difference > 0)
|
|
241
|
-
byId.set(
|
|
259
|
+
byId.set(key, receipt);
|
|
242
260
|
}
|
|
243
261
|
return [...byId.values()];
|
|
244
262
|
}
|
|
@@ -288,13 +306,13 @@ export async function readFrontendSessionBudgetReceipts(runDir, nodeId = "fronte
|
|
|
288
306
|
// Keep a visible incomplete receipt so canary aggregation cannot turn a
|
|
289
307
|
// truncated artifact into an apparent zero-cost improvement.
|
|
290
308
|
receipts.push({
|
|
291
|
-
schemaVersion:
|
|
292
|
-
policyVersion: "frontend-session-observation-
|
|
293
|
-
identity: { sessionId: `invalid:${file}`, phase: "unknown", scopeIds: [], taskId: null, runId: null, nodeId, attempt: null, sourceDigest: null, contractDigest: null },
|
|
309
|
+
schemaVersion: 2,
|
|
310
|
+
policyVersion: "frontend-session-observation-v2",
|
|
311
|
+
identity: { sessionId: `invalid:${file}`, phase: "unknown", executionKind: "model", dispatchReason: "initial", scopeIds: [], taskId: null, runId: null, nodeId, attempt: null, sourceDigest: null, contractDigest: null },
|
|
294
312
|
startedAt: new Date(0).toISOString(), finishedAt: null, completion: "incomplete",
|
|
295
313
|
planned: { prompt: { sha256: "", chars: 0, bytes: 0, estimatedTokens: 0 }, userMessage: { sha256: "", chars: 0, bytes: 0, estimatedTokens: 0 }, toolSchemas: null, estimatorVersion: "unknown", capabilities: { ...frontendModelCapabilities(undefined), model: null }, unmeasuredComponents: [] },
|
|
296
314
|
actual: { ...collectFrontendUsage([]) }, attempts: [],
|
|
297
|
-
progress: { memoryCommittedDelta: null, durableCommittedDelta: null, failedRecords: null, duplicateRecords: null, durableSubmissions: null, durableReplacements: null, successfulReadBytes: null, duplicateReadBytes: null }, terminal: null,
|
|
315
|
+
progress: { finalizeRejections: null, memoryCommittedDelta: null, durableCommittedDelta: null, failedRecords: null, duplicateRecords: null, durableSubmissions: null, durableReplacements: null, successfulReadBytes: null, duplicateReadBytes: null }, terminal: null,
|
|
298
316
|
});
|
|
299
317
|
}
|
|
300
318
|
}
|
|
@@ -302,7 +320,8 @@ export async function readFrontendSessionBudgetReceipts(runDir, nodeId = "fronte
|
|
|
302
320
|
}
|
|
303
321
|
export function summarizeFrontendPlanCanary(receipts) {
|
|
304
322
|
const unique = uniqueFrontendSessions(receipts);
|
|
305
|
-
const
|
|
323
|
+
const model = unique.filter(r => r.identity.executionKind === "model");
|
|
324
|
+
const complete = model.length > 0 && unique.every(receipt => isCanaryReceipt(receipt) && receipt.completion === "complete");
|
|
306
325
|
const known = (values, sum) => {
|
|
307
326
|
if (!complete || values.length === 0)
|
|
308
327
|
return null;
|
|
@@ -310,14 +329,17 @@ export function summarizeFrontendPlanCanary(receipts) {
|
|
|
310
329
|
return present.length === values.length ? sum(present) : null;
|
|
311
330
|
};
|
|
312
331
|
return {
|
|
313
|
-
planModelRequests: known(
|
|
314
|
-
planSessions: complete ?
|
|
332
|
+
planModelRequests: known(model.map((receipt) => receipt.actual.requestCount), (items) => items.reduce((total, value) => total + value, 0)),
|
|
333
|
+
planSessions: complete ? model.length : null,
|
|
315
334
|
// Receipt envelope includes gaps and overlapping sessions only once. The
|
|
316
335
|
// run reader below replaces this with the authoritative node interval.
|
|
317
336
|
planWallClockMs: complete ? Math.max(...unique.map(r => Date.parse(r.finishedAt))) - Math.min(...unique.map(r => Date.parse(r.startedAt))) : null,
|
|
318
|
-
planTokens: known(
|
|
319
|
-
toolRejections: known(
|
|
320
|
-
|
|
337
|
+
planTokens: known(model.map((receipt) => receipt.actual.totalTokens), (items) => items.reduce((total, value) => total + value, 0)),
|
|
338
|
+
toolRejections: known(model.map((receipt) => receipt.progress.failedRecords), (items) => items.reduce((total, value) => total + value, 0)),
|
|
339
|
+
factReplacements: known(model.map(r => r.progress.durableReplacements), items => items.reduce((a, b) => a + b, 0)),
|
|
340
|
+
finalizeRejections: known(unique.map(r => r.progress.finalizeRejections), items => items.reduce((a, b) => a + b, 0)),
|
|
341
|
+
correctionSessions: complete ? model.filter(r => r.identity.dispatchReason === "correction").length : null,
|
|
342
|
+
transportRetries: complete ? model.reduce((total, receipt) => total + Math.max(0, receipt.attempts.length - 1), 0) : null,
|
|
321
343
|
};
|
|
322
344
|
}
|
|
323
345
|
/** Validate the fields consumed by aggregation before accessing nested values.
|
|
@@ -325,11 +347,15 @@ export function summarizeFrontendPlanCanary(receipts) {
|
|
|
325
347
|
function isCanaryReceipt(value) {
|
|
326
348
|
const r = record(value), identity = record(r.identity), actual = record(r.actual), progress = record(r.progress);
|
|
327
349
|
const nullableCount = (v) => v === null || (typeof v === "number" && Number.isFinite(v) && Number.isInteger(v) && v >= 0);
|
|
328
|
-
return r.schemaVersion ===
|
|
350
|
+
return r.schemaVersion === 2 && r.policyVersion === "frontend-session-observation-v2"
|
|
351
|
+
&& ["model", "runtime"].includes(String(identity.executionKind))
|
|
352
|
+
&& ["initial", "correction"].includes(String(identity.dispatchReason))
|
|
353
|
+
&& typeof identity.phase === "string" && Array.isArray(identity.scopeIds)
|
|
329
354
|
&& typeof identity.sessionId === "string" && identity.sessionId.length > 0
|
|
330
355
|
&& typeof r.startedAt === "string" && Number.isFinite(Date.parse(r.startedAt))
|
|
331
356
|
&& (r.completion === "incomplete" || (r.completion === "complete" && typeof r.finishedAt === "string" && Date.parse(r.finishedAt) >= Date.parse(r.startedAt)))
|
|
332
357
|
&& Array.isArray(r.attempts) && r.attempts.every(a => a !== null && typeof a === "object")
|
|
358
|
+
&& nullableCount(progress.durableReplacements) && nullableCount(progress.finalizeRejections)
|
|
333
359
|
&& nullableCount(actual.requestCount) && nullableCount(actual.totalTokens) && nullableCount(progress.failedRecords);
|
|
334
360
|
}
|
|
335
361
|
/** Read one frozen canary run. The optional artifact may add only run-level
|
|
@@ -337,27 +363,14 @@ function isCanaryReceipt(value) {
|
|
|
337
363
|
* request usage and tool rejections always come from session receipts. */
|
|
338
364
|
export async function readFrontendPlanCanaryRunMetrics(runDir, nodeId = "frontend-plan-pi") {
|
|
339
365
|
const sessionMetrics = summarizeFrontendPlanCanary(await readFrontendSessionBudgetReceipts(runDir, nodeId));
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
|
|
344
|
-
supplemental = parsed;
|
|
345
|
-
}
|
|
346
|
-
}
|
|
347
|
-
catch {
|
|
348
|
-
// Missing or malformed supplemental evidence remains unavailable.
|
|
349
|
-
}
|
|
350
|
-
const optionalNumber = (key) => {
|
|
351
|
-
const value = supplemental[key];
|
|
352
|
-
return typeof value === "number" && Number.isFinite(value) && value >= 0
|
|
353
|
-
? value
|
|
354
|
-
: null;
|
|
355
|
-
};
|
|
366
|
+
// A freely editable metrics JSON is not execution or adjudication evidence.
|
|
367
|
+
// Root evaluation must supply validated lineage and independent acceptance;
|
|
368
|
+
// until then the Plan projection can pass only its own local gate.
|
|
356
369
|
return {
|
|
357
370
|
...sessionMetrics,
|
|
358
|
-
dagWallClockMs:
|
|
359
|
-
designReviewRejectRate:
|
|
360
|
-
finalReviewDefectRate:
|
|
361
|
-
recoveryCount:
|
|
371
|
+
dagWallClockMs: null,
|
|
372
|
+
designReviewRejectRate: null,
|
|
373
|
+
finalReviewDefectRate: null,
|
|
374
|
+
recoveryCount: null,
|
|
362
375
|
};
|
|
363
376
|
}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import { AsyncLocalStorage } from "node:async_hooks";
|
|
2
|
+
const context = new AsyncLocalStorage();
|
|
3
|
+
/** Async-local identity only; no mutable budget or usage is stored here. */
|
|
4
|
+
export function withFrontendSessionIdentity(identity, execute) {
|
|
5
|
+
return context.run(structuredClone(identity), execute);
|
|
6
|
+
}
|
|
7
|
+
export function currentFrontendSessionIdentity() {
|
|
8
|
+
const identity = context.getStore();
|
|
9
|
+
return identity && structuredClone(identity);
|
|
10
|
+
}
|