@tangle-network/agent-eval 0.174.0 → 0.176.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +37 -0
- package/README.md +1 -1
- package/dist/adapters/http.d.ts +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js +374 -0
- package/dist/agent-profile-cell-0gSi5ffD.js.map +1 -0
- package/dist/analyst/index.d.ts +3 -3
- package/dist/analyst/index.js +4 -4
- package/dist/{benchmark-command-mZIlR-ra.js → benchmark-command-yPqjcZnC.js} +7 -7
- package/dist/{benchmark-command-mZIlR-ra.js.map → benchmark-command-yPqjcZnC.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +3 -3
- package/dist/campaign/index.js +9 -9
- package/dist/{campaign-BzMSCejE.js → campaign-85igdlgG.js} +12 -12
- package/dist/{campaign-BzMSCejE.js.map → campaign-85igdlgG.js.map} +1 -1
- package/dist/campaign-evidence-D8DBLqLI.js +2083 -0
- package/dist/campaign-evidence-D8DBLqLI.js.map +1 -0
- package/dist/{opencode-sqlite-eK6HW6dr.js → claude-jsonl-CxZZrDJ3.js} +9 -149
- package/dist/claude-jsonl-CxZZrDJ3.js.map +1 -0
- package/dist/cli.js +9 -2
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +4 -4
- package/dist/contract/index.js +9 -9
- package/dist/{default-registry-CrAp0pYq.js → default-registry-DBqVI4pq.js} +2 -2
- package/dist/{default-registry-CrAp0pYq.js.map → default-registry-DBqVI4pq.js.map} +1 -1
- package/dist/{define-agent-eval-V1jQyCDR.d.ts → define-agent-eval-CCbl8k2E.d.ts} +11 -4
- package/dist/define-agent-eval-CCbl8k2E.d.ts.map +1 -0
- package/dist/{define-agent-eval-ox5McL6e.js → define-agent-eval-DEMsu5eA.js} +54 -36
- package/dist/define-agent-eval-DEMsu5eA.js.map +1 -0
- package/dist/{dspy-rlm-engine-Caz2pl4L.js → dspy-rlm-engine-DqjER2sV.js} +2 -2
- package/dist/{dspy-rlm-engine-Caz2pl4L.js.map → dspy-rlm-engine-DqjER2sV.js.map} +1 -1
- package/dist/{eval-campaign-BeAjdhzC.js → eval-campaign-Cs-7MiCs.js} +4 -5
- package/dist/{eval-campaign-BeAjdhzC.js.map → eval-campaign-Cs-7MiCs.js.map} +1 -1
- package/dist/experiment/index.d.ts +3 -68
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +6 -128
- package/dist/experiment/index.js.map +1 -1
- package/dist/{attestation-XSUpbc4o.js → experiment-tracker-BKEumQug.js} +2 -96
- package/dist/experiment-tracker-BKEumQug.js.map +1 -0
- package/dist/{attestation-c1QvaBdX.d.ts → experiment-tracker-CNwqCZFD.d.ts} +2 -78
- package/dist/experiment-tracker-CNwqCZFD.d.ts.map +1 -0
- package/dist/{external-optimizer-process-CxnFL1hd.js → external-optimizer-process-Dlz8YxrT.js} +3 -3
- package/dist/{external-optimizer-process-CxnFL1hd.js.map → external-optimizer-process-Dlz8YxrT.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-CQi27uEI.js → external-optimizer-subprocess-q3VzlGAO.js} +2 -2
- package/dist/{external-optimizer-subprocess-CQi27uEI.js.map → external-optimizer-subprocess-q3VzlGAO.js.map} +1 -1
- package/dist/{index-Bn-nlnSV.d.ts → index-BAAiSF3_.d.ts} +2 -2
- package/dist/{index-Bn-nlnSV.d.ts.map → index-BAAiSF3_.d.ts.map} +1 -1
- package/dist/{index-DKXuBPXf.d.ts → index-Bg6OT2Dd.d.ts} +23 -10
- package/dist/{index-DKXuBPXf.d.ts.map → index-Bg6OT2Dd.d.ts.map} +1 -1
- package/dist/{index-BTrx5s8m.d.ts → index-DBkcm_9H.d.ts} +4 -4
- package/dist/{index-BTrx5s8m.d.ts.map → index-DBkcm_9H.d.ts.map} +1 -1
- package/dist/{index-D-UdhAmg.d.ts → index-u0d1Jp4F.d.ts} +4 -2
- package/dist/{index-D-UdhAmg.d.ts.map → index-u0d1Jp4F.d.ts.map} +1 -1
- package/dist/index.d.ts +5 -5
- package/dist/index.js +15 -16
- package/dist/index.js.map +1 -1
- package/dist/{integrity-BWywb34E.js → integrity-DsHWCebQ.js} +11 -435
- package/dist/integrity-DsHWCebQ.js.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-PIfjCbKn.js → ledger-core-Cs9f7385.js} +60 -47
- package/dist/{ledger-core-PIfjCbKn.js.map → ledger-core-Cs9f7385.js.map} +1 -1
- package/dist/{llm-judge-DmNaBrXB.js → llm-judge-DliimmRb.js} +994 -1517
- package/dist/llm-judge-DliimmRb.js.map +1 -0
- package/dist/{mint-vWOdD8Ae.js → mint-Cc1_zwRQ.js} +2 -2
- package/dist/{mint-vWOdD8Ae.js.map → mint-Cc1_zwRQ.js.map} +1 -1
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-CNw3vubS.js +145 -0
- package/dist/opencode-sqlite-CNw3vubS.js.map +1 -0
- package/dist/{produced-state-B8mw6zj9.js → produced-state-DrMqa2HD.js} +3 -2
- package/dist/{produced-state-B8mw6zj9.js.map → produced-state-DrMqa2HD.js.map} +1 -1
- package/dist/profile-cell.js +1 -268
- package/dist/{promotion-policy-LY9mVQ7W.js → promotion-policy-DWOm70gx.js} +2 -2
- package/dist/{promotion-policy-LY9mVQ7W.js.map → promotion-policy-DWOm70gx.js.map} +1 -1
- package/dist/{release-confidence-BsGEg_xg.js → release-confidence-BcGCclTB.js} +2 -2
- package/dist/{release-confidence-BsGEg_xg.js.map → release-confidence-BcGCclTB.js.map} +1 -1
- package/dist/report-command-DKlXfU5r.js +1528 -0
- package/dist/report-command-DKlXfU5r.js.map +1 -0
- package/dist/reporting.js +2 -2
- package/dist/{reward-hacking-CKW4teig.js → reward-hacking-D0XwhVWE.js} +2 -215
- package/dist/reward-hacking-D0XwhVWE.js.map +1 -0
- package/dist/rl.js +5 -4
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.js +4 -3
- package/dist/{rollout-C-znbbYg.js → rollout-DmoJVqrF.js} +4 -3
- package/dist/{rollout-C-znbbYg.js.map → rollout-DmoJVqrF.js.map} +1 -1
- package/dist/run-record-CR63CpHK.js +216 -0
- package/dist/run-record-CR63CpHK.js.map +1 -0
- package/dist/{run-record-ZIsR9Fif.js → run-record-DQpSf7t-.js} +2 -2
- package/dist/{run-record-ZIsR9Fif.js.map → run-record-DQpSf7t-.js.map} +1 -1
- package/dist/{semantic-concept-judge-E3s_fEjB.js → semantic-concept-judge-Dw-f7TEs.js} +3 -3
- package/dist/{semantic-concept-judge-E3s_fEjB.js.map → semantic-concept-judge-Dw-f7TEs.js.map} +1 -1
- package/dist/{sequential-B51qAYE4.js → sequential-B5gXgcyp.js} +3 -3
- package/dist/{sequential-B51qAYE4.js.map → sequential-B5gXgcyp.js.map} +1 -1
- package/dist/{skillopt-optimization-method-f7399oGb.js → skillopt-optimization-method-CV7go7ex.js} +6 -749
- package/dist/skillopt-optimization-method-CV7go7ex.js.map +1 -0
- package/dist/{statistical-heldout-Cqb73yE9.d.ts → statistical-heldout-Z9NROFFS.d.ts} +156 -3
- package/dist/statistical-heldout-Z9NROFFS.d.ts.map +1 -0
- package/dist/{summary-report-Bgh8CpNK.js → summary-report-B16xy9Kd.js} +2 -2
- package/dist/{summary-report-Bgh8CpNK.js.map → summary-report-B16xy9Kd.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +71 -6
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +6 -1357
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/terminal-record-Ce9_UjRz.js +539 -0
- package/dist/terminal-record-Ce9_UjRz.js.map +1 -0
- package/dist/traces.js +1 -1
- package/dist/{types-CoPUTiXb.d.ts → types-vUdAx2Cj.d.ts} +65 -3
- package/dist/types-vUdAx2Cj.d.ts.map +1 -0
- package/docs/public-api.md +62 -39
- package/docs/search-history-receipts.md +48 -1
- package/package.json +1 -1
- package/dist/attestation-XSUpbc4o.js.map +0 -1
- package/dist/attestation-c1QvaBdX.d.ts.map +0 -1
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +0 -1
- package/dist/define-agent-eval-ox5McL6e.js.map +0 -1
- package/dist/integrity-BWywb34E.js.map +0 -1
- package/dist/llm-judge-DmNaBrXB.js.map +0 -1
- package/dist/opencode-sqlite-eK6HW6dr.js.map +0 -1
- package/dist/power-preflight-CFXm0Vjo.js +0 -502
- package/dist/power-preflight-CFXm0Vjo.js.map +0 -1
- package/dist/pre-registration-D94b7Of5.js +0 -110
- package/dist/pre-registration-D94b7Of5.js.map +0 -1
- package/dist/profile-cell.js.map +0 -1
- package/dist/reward-hacking-CKW4teig.js.map +0 -1
- package/dist/skillopt-optimization-method-f7399oGb.js.map +0 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +0 -1
- package/dist/types-CoPUTiXb.d.ts.map +0 -1
|
@@ -1,15 +1,16 @@
|
|
|
1
|
-
import { i as JudgeError, s as ValidationError
|
|
1
|
+
import { i as JudgeError, s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
2
|
import { a as hashCanonical, i as compareCodeUnits, r as canonicalString } from "./canonical-DPyQ_rpt.js";
|
|
3
3
|
import { o as summarizeNumberSeries, s as weightedComposite, t as confidenceInterval } from "./descriptive-1V17A-qa.js";
|
|
4
4
|
import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
|
|
5
|
-
import { i as modelHasSnapshot } from "./run-record-
|
|
5
|
+
import { i as modelHasSnapshot } from "./run-record-DQpSf7t-.js";
|
|
6
6
|
import { n as contentHash } from "./verdict-cache-B3eCVQtY.js";
|
|
7
|
-
import {
|
|
7
|
+
import { o as projectCampaignCellQuality, t as campaignCellCostProvenance } from "./run-record-CR63CpHK.js";
|
|
8
8
|
import { i as CostLedger, t as CostAccountingIncompleteError } from "./cost-ledger-B1qx30B4.js";
|
|
9
|
-
import {
|
|
10
|
-
import {
|
|
9
|
+
import { $ as campaignScenarioIdentity, C as campaignMeanCompositeOrNull, G as surfaceHashMatches, H as surfaceContentHash, K as BackendIntegrityError, M as heldoutSignificance, N as pairHoldout, Q as campaignCoverage, S as campaignMeanComposite, U as surfaceDispatchRef, V as renderSurfaceDiff, W as surfaceHash, X as assertCampaignSplitIdentity, Y as assertCampaignDesign, Z as assertCompleteCampaign, b as assertFiniteRankKey, et as campaignSplitDigest, j as dimensionRegressions, nt as formatCoverageFailures, t as createCampaignEvidenceReceipt, w as compareRankKeys, x as campaignBreakdown } from "./campaign-evidence-D8DBLqLI.js";
|
|
10
|
+
import { d as mapConcurrent, n as replayLedgerText, t as FileLedgerJournal } from "./ledger-core-Cs9f7385.js";
|
|
11
|
+
import { D as SearchLedgerIntegrityError, E as SearchLedgerError, S as fsCampaignStorage, T as SearchLedgerConflictError, g as isRecord, h as isExternalTextCandidate, w as SEARCH_LEDGER_FILE_CONTEXT, x as createRunCostLedger } from "./external-optimizer-subprocess-q3VzlGAO.js";
|
|
11
12
|
import { t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
12
|
-
import {
|
|
13
|
+
import { t as detectRewardHacking } from "./reward-hacking-D0XwhVWE.js";
|
|
13
14
|
import { n as assertProposalFindings, t as combineAbortSignals } from "./abort-signal-CtzAM_sJ.js";
|
|
14
15
|
import { l as deepFreezeCanonicalJson } from "./types-DQ0e2E7y.js";
|
|
15
16
|
import { d as stripFencedJson, o as costReceiptFromLlm, s as costReceiptFromLlmError, u as maximumChargeForLlmRequest } from "./llm-client-CGlSi8sb.js";
|
|
@@ -19,6 +20,7 @@ import { z } from "zod";
|
|
|
19
20
|
import { basename, isAbsolute, join } from "node:path";
|
|
20
21
|
import { homedir, tmpdir } from "node:os";
|
|
21
22
|
import { execSync } from "node:child_process";
|
|
23
|
+
import { fileURLToPath } from "node:url";
|
|
22
24
|
//#region src/campaign/campaign-manifest.ts
|
|
23
25
|
/**
|
|
24
26
|
* Campaign identity: the manifest hash over (scenarios, judges, dispatch,
|
|
@@ -286,251 +288,6 @@ function invalidCachedCellsError(cells) {
|
|
|
286
288
|
return new CostAccountingIncompleteError(`runCampaign: cached cell(s) require explicit paid re-dispatch: ${cells.map((cell) => `${cell.cellId} (${cell.reason}${cell.detail ? `: ${cell.detail}` : ""})`).join(", ")}; refusing to begin campaign. Inspect planCampaignRun, then set rerunInvalidCachedCells: true to rerun only these cells while retaining valid caches, or resumable: false when a full rerun is intended.`);
|
|
287
289
|
}
|
|
288
290
|
//#endregion
|
|
289
|
-
//#region src/campaign/coverage.ts
|
|
290
|
-
/** Reject campaign designs whose denominator cannot be identified exactly. */
|
|
291
|
-
function assertCampaignDesign(scenarios, reps) {
|
|
292
|
-
if (!Number.isSafeInteger(reps) || reps < 1) throw new Error("campaign design requires reps to be a positive safe integer");
|
|
293
|
-
const scenarioIds = /* @__PURE__ */ new Set();
|
|
294
|
-
for (const scenario of scenarios) {
|
|
295
|
-
if (typeof scenario.id !== "string" || scenario.id.trim().length === 0) throw new Error("campaign design requires every scenario to have a non-empty id");
|
|
296
|
-
if (scenarioIds.has(scenario.id)) throw new Error(`campaign design contains duplicate scenario id '${scenario.id}'`);
|
|
297
|
-
if (typeof scenario.kind !== "string" || scenario.kind.trim().length === 0) throw new Error("campaign design requires every scenario to have a non-empty kind");
|
|
298
|
-
if (scenario.seedGroup !== void 0 && (typeof scenario.seedGroup !== "string" || scenario.seedGroup.trim().length === 0)) throw new Error("campaign design requires seedGroup to be a non-empty string when set");
|
|
299
|
-
scenarioIds.add(scenario.id);
|
|
300
|
-
}
|
|
301
|
-
}
|
|
302
|
-
/** Redacted but independently verifiable identity of one complete scenario. */
|
|
303
|
-
function campaignScenarioIdentity(scenario) {
|
|
304
|
-
assertCampaignDesign([scenario], 1);
|
|
305
|
-
return {
|
|
306
|
-
id: scenario.id,
|
|
307
|
-
kind: scenario.kind,
|
|
308
|
-
scenarioDigest: `sha256:${contentHash(scenario)}`
|
|
309
|
-
};
|
|
310
|
-
}
|
|
311
|
-
/** Canonical split identity reconstructed from redacted scenario identities. */
|
|
312
|
-
function campaignSplitDigestFromIdentities(scenarios, reps) {
|
|
313
|
-
assertCampaignDesign(scenarios, reps);
|
|
314
|
-
for (const scenario of scenarios) if (!/^sha256:[a-f0-9]{64}$/.test(scenario.scenarioDigest)) throw new Error(`campaign scenario '${scenario.id}' has an invalid digest`);
|
|
315
|
-
return `sha256:${contentHash({
|
|
316
|
-
schema: "tangle.campaign-split",
|
|
317
|
-
scenarios: scenarios.map(({ id, kind, scenarioDigest }) => ({
|
|
318
|
-
id,
|
|
319
|
-
kind,
|
|
320
|
-
scenarioDigest
|
|
321
|
-
})),
|
|
322
|
-
reps
|
|
323
|
-
})}`;
|
|
324
|
-
}
|
|
325
|
-
/** Canonical identity of the exact scenario payloads and replicate count. */
|
|
326
|
-
function campaignSplitDigest(scenarios, reps) {
|
|
327
|
-
assertCampaignDesign(scenarios, reps);
|
|
328
|
-
return campaignSplitDigestFromIdentities(scenarios.map(campaignScenarioIdentity), reps);
|
|
329
|
-
}
|
|
330
|
-
/** Refuse a campaign whose retained task identities contradict its split digest. */
|
|
331
|
-
function assertCampaignSplitIdentity(scenarios, reps, splitDigest) {
|
|
332
|
-
if (campaignSplitDigestFromIdentities(scenarios, reps) !== splitDigest) throw new Error("campaign split digest does not match its retained scenario identities");
|
|
333
|
-
}
|
|
334
|
-
/** Exact designed-denominator receipt for one campaign. */
|
|
335
|
-
function campaignCoverage(cells, scenarios, reps, requireJudgeScore) {
|
|
336
|
-
assertCampaignDesign(scenarios, reps);
|
|
337
|
-
const expectedCellIds = designedCellIds(scenarios, reps);
|
|
338
|
-
const cellsById = /* @__PURE__ */ new Map();
|
|
339
|
-
for (const cell of cells) {
|
|
340
|
-
const matches = cellsById.get(cell.cellId) ?? [];
|
|
341
|
-
matches.push(cell);
|
|
342
|
-
cellsById.set(cell.cellId, matches);
|
|
343
|
-
}
|
|
344
|
-
const scorableCellIds = [];
|
|
345
|
-
const unscorableCells = [];
|
|
346
|
-
for (const cellId of expectedCellIds) {
|
|
347
|
-
const matches = cellsById.get(cellId) ?? [];
|
|
348
|
-
if (matches.length === 0) {
|
|
349
|
-
unscorableCells.push({
|
|
350
|
-
cellId,
|
|
351
|
-
reason: "missing campaign cell"
|
|
352
|
-
});
|
|
353
|
-
continue;
|
|
354
|
-
}
|
|
355
|
-
if (matches.length > 1) {
|
|
356
|
-
unscorableCells.push({
|
|
357
|
-
cellId,
|
|
358
|
-
reason: `duplicate campaign cell (${matches.length})`
|
|
359
|
-
});
|
|
360
|
-
continue;
|
|
361
|
-
}
|
|
362
|
-
const cell = matches[0];
|
|
363
|
-
const scoreEntries = Object.entries(cell.judgeScores);
|
|
364
|
-
const successfulScores = scoreEntries.map(([, score]) => score).filter((score) => score.failed !== true && Number.isFinite(score.composite));
|
|
365
|
-
const nonFiniteScores = scoreEntries.filter(([, score]) => score.failed !== true && (!Number.isFinite(score.composite) || Object.values(score.dimensions).some((value) => !Number.isFinite(value))));
|
|
366
|
-
const reasons = [];
|
|
367
|
-
if (cell.error) reasons.push(cell.error);
|
|
368
|
-
if (cell.artifact === null || cell.artifact === void 0) reasons.push("missing artifact");
|
|
369
|
-
if (!cell.error && requireJudgeScore && successfulScores.length === 0) reasons.push("no successful finite judge score");
|
|
370
|
-
if (scoreEntries.some(([, score]) => score.failed === true)) reasons.push("judge score marked failed");
|
|
371
|
-
const failedPanelJudges = [...new Set(scoreEntries.flatMap(([, score]) => score.failedJudges ?? []))].sort();
|
|
372
|
-
if (failedPanelJudges.length > 0) reasons.push(`judge panel incomplete: ${failedPanelJudges.join(", ")}`);
|
|
373
|
-
if (nonFiniteScores.length > 0) reasons.push(`non-finite judge score: ${nonFiniteScores.map(([name]) => name).sort().join(", ")}`);
|
|
374
|
-
if (reasons.length > 0) unscorableCells.push({
|
|
375
|
-
cellId,
|
|
376
|
-
reason: reasons.join("; ")
|
|
377
|
-
});
|
|
378
|
-
else scorableCellIds.push(cellId);
|
|
379
|
-
}
|
|
380
|
-
const expected = new Set(expectedCellIds);
|
|
381
|
-
for (const cell of cells) {
|
|
382
|
-
if (cell.cellId !== `${cell.scenarioId}:${cell.rep}`) {
|
|
383
|
-
unscorableCells.push({
|
|
384
|
-
cellId: cell.cellId,
|
|
385
|
-
reason: "campaign cell id does not match scenario id and rep"
|
|
386
|
-
});
|
|
387
|
-
continue;
|
|
388
|
-
}
|
|
389
|
-
if (!expected.has(cell.cellId)) unscorableCells.push({
|
|
390
|
-
cellId: cell.cellId,
|
|
391
|
-
reason: "unexpected campaign cell"
|
|
392
|
-
});
|
|
393
|
-
}
|
|
394
|
-
return {
|
|
395
|
-
complete: unscorableCells.length === 0 && scorableCellIds.length === expectedCellIds.length,
|
|
396
|
-
expectedCellIds,
|
|
397
|
-
scorableCellIds,
|
|
398
|
-
unscorableCells
|
|
399
|
-
};
|
|
400
|
-
}
|
|
401
|
-
function formatCoverageFailures(coverage) {
|
|
402
|
-
const shown = coverage.unscorableCells.slice(0, 3).map((cell) => `${cell.cellId}: ${cell.reason}`).join("; ");
|
|
403
|
-
const remainder = coverage.unscorableCells.length - Math.min(3, coverage.unscorableCells.length);
|
|
404
|
-
return remainder > 0 ? `${shown}; +${remainder} more` : shown || "unknown coverage failure";
|
|
405
|
-
}
|
|
406
|
-
function designedCellIds(scenarios, reps) {
|
|
407
|
-
const ids = [];
|
|
408
|
-
for (const scenario of scenarios) for (let rep = 0; rep < reps; rep++) ids.push(`${scenario.id}:${rep}`);
|
|
409
|
-
return ids;
|
|
410
|
-
}
|
|
411
|
-
/** Require the complete designed denominator before a final comparison. */
|
|
412
|
-
function assertCompleteCampaign(campaign, scenarios, reps, requireJudgeScore, label) {
|
|
413
|
-
const coverage = campaignCoverage(campaign.cells, scenarios, reps, requireJudgeScore);
|
|
414
|
-
if (!coverage.complete) throw new Error(`${label} is incomplete (${coverage.scorableCellIds.length}/${coverage.expectedCellIds.length} designed cells scorable) — ${formatCoverageFailures(coverage)}. Refusing to compare unequal results.`);
|
|
415
|
-
}
|
|
416
|
-
//#endregion
|
|
417
|
-
//#region src/integrity/backend-integrity.ts
|
|
418
|
-
/**
|
|
419
|
-
* Error thrown when an integrity assertion fails. Caller can pattern-match
|
|
420
|
-
* by `code === 'AGENT_EVAL_BACKEND_STUB'` to differentiate from other
|
|
421
|
-
* errors.
|
|
422
|
-
*/
|
|
423
|
-
var BackendIntegrityError = class extends AgentEvalError {
|
|
424
|
-
report;
|
|
425
|
-
constructor(message, report) {
|
|
426
|
-
super("backend_integrity", message);
|
|
427
|
-
this.report = report;
|
|
428
|
-
}
|
|
429
|
-
};
|
|
430
|
-
/**
|
|
431
|
-
* Inspect a batch of RunRecords and return an integrity report. Pure
|
|
432
|
-
* function — no I/O, no logging. The caller decides what to do with the
|
|
433
|
-
* verdict (print warning, throw, gate CI, etc.).
|
|
434
|
-
*/
|
|
435
|
-
function summarizeBackendIntegrity(records) {
|
|
436
|
-
return summarizeBackendUsage(records.map((record) => ({
|
|
437
|
-
inputTokens: record.tokenUsage.input,
|
|
438
|
-
outputTokens: record.tokenUsage.output,
|
|
439
|
-
costUsd: record.costUsd
|
|
440
|
-
})));
|
|
441
|
-
}
|
|
442
|
-
/** Inspect settled agent calls from the canonical cost ledger. */
|
|
443
|
-
function summarizeAgentReceiptIntegrity(receipts) {
|
|
444
|
-
return summarizeBackendUsage(receipts.filter((receipt) => receipt.channel === "agent").map((receipt) => ({
|
|
445
|
-
inputTokens: receipt.inputTokens,
|
|
446
|
-
outputTokens: receipt.outputTokens,
|
|
447
|
-
costUsd: receipt.costUsd
|
|
448
|
-
})));
|
|
449
|
-
}
|
|
450
|
-
function summarizeBackendUsage(records) {
|
|
451
|
-
const totalRecords = records.length;
|
|
452
|
-
let stubRecords = 0;
|
|
453
|
-
let realRecords = 0;
|
|
454
|
-
let uncostedRecords = 0;
|
|
455
|
-
let totalInputTokens = 0;
|
|
456
|
-
let totalOutputTokens = 0;
|
|
457
|
-
let totalCostUsd = 0;
|
|
458
|
-
for (const rec of records) {
|
|
459
|
-
totalInputTokens += rec.inputTokens;
|
|
460
|
-
totalOutputTokens += rec.outputTokens;
|
|
461
|
-
totalCostUsd += rec.costUsd ?? 0;
|
|
462
|
-
if (rec.inputTokens === 0 && rec.outputTokens === 0) stubRecords++;
|
|
463
|
-
else realRecords++;
|
|
464
|
-
if (rec.outputTokens > 0 && (rec.costUsd === null || rec.costUsd === 0)) uncostedRecords++;
|
|
465
|
-
}
|
|
466
|
-
const verdict = totalRecords === 0 ? "stub" : stubRecords === totalRecords ? "stub" : stubRecords === 0 ? "real" : "mixed";
|
|
467
|
-
const diagnosis = buildDiagnosis({
|
|
468
|
-
totalRecords,
|
|
469
|
-
stubRecords,
|
|
470
|
-
realRecords,
|
|
471
|
-
uncostedRecords,
|
|
472
|
-
totalInputTokens,
|
|
473
|
-
totalOutputTokens,
|
|
474
|
-
totalCostUsd,
|
|
475
|
-
verdict
|
|
476
|
-
});
|
|
477
|
-
return {
|
|
478
|
-
totalRecords,
|
|
479
|
-
stubRecords,
|
|
480
|
-
realRecords,
|
|
481
|
-
uncostedRecords,
|
|
482
|
-
totalInputTokens,
|
|
483
|
-
totalOutputTokens,
|
|
484
|
-
totalCostUsd,
|
|
485
|
-
verdict,
|
|
486
|
-
diagnosis
|
|
487
|
-
};
|
|
488
|
-
}
|
|
489
|
-
function buildDiagnosis(r) {
|
|
490
|
-
if (r.totalRecords === 0) return "no records — eval produced zero runs; backend likely failed before first turn";
|
|
491
|
-
if (r.verdict === "stub") return [
|
|
492
|
-
`all ${r.totalRecords} records have zero token usage — the LLM backend was never called.`,
|
|
493
|
-
"common causes: --backend sandbox without a sandbox bridge running; stub model returning hard-coded strings;",
|
|
494
|
-
"auth misconfigured so requests were silently dropped before the LLM. Re-run with --backend tcloud and TANGLE_API_KEY set,",
|
|
495
|
-
"or boot the cli-bridge / sandbox before invoking the eval."
|
|
496
|
-
].join(" ");
|
|
497
|
-
if (r.verdict === "mixed") {
|
|
498
|
-
const pct = (r.stubRecords / r.totalRecords * 100).toFixed(0);
|
|
499
|
-
return [
|
|
500
|
-
`${r.stubRecords}/${r.totalRecords} records (${pct}%) have zero token usage — the backend partially failed.`,
|
|
501
|
-
"common causes: rate-limit cascade (429s after the first N personas);",
|
|
502
|
-
"transient auth expiry mid-run; provider outage. Treat the affected records as missing data, not agent failures."
|
|
503
|
-
].join(" ");
|
|
504
|
-
}
|
|
505
|
-
if (r.uncostedRecords > 0) {
|
|
506
|
-
const pct = (r.uncostedRecords / r.totalRecords * 100).toFixed(0);
|
|
507
|
-
return [
|
|
508
|
-
`${r.totalRecords} records with real LLM activity (in=${r.totalInputTokens}, out=${r.totalOutputTokens} tokens).`,
|
|
509
|
-
`${r.uncostedRecords} (${pct}%) have output tokens but costUsd=0. Two distinct roots:`,
|
|
510
|
-
"(a) cost ledger mis-wired — no usage propagation from the runtime stream into RunRecord; or",
|
|
511
|
-
"(b) the model is unpriced at the source (sandbox/router returned $0 despite real tokens).",
|
|
512
|
-
"For (b), price the measured tokens against the substrate table (estimateCost) instead of leaving $0."
|
|
513
|
-
].join(" ");
|
|
514
|
-
}
|
|
515
|
-
return `${r.totalRecords} records with real LLM activity (in=${r.totalInputTokens}, out=${r.totalOutputTokens} tokens, $${r.totalCostUsd.toFixed(4)}).`;
|
|
516
|
-
}
|
|
517
|
-
/**
|
|
518
|
-
* Throw BackendIntegrityError if the verdict is 'stub' — i.e. every record
|
|
519
|
-
* shows zero LLM activity. Non-strict callers can pass `{ allowMixed: false }`
|
|
520
|
-
* to also reject mixed verdicts (recommended for CI gates).
|
|
521
|
-
*
|
|
522
|
-
* Real backends pass through silently.
|
|
523
|
-
*/
|
|
524
|
-
function assertRealBackend(records, opts = {}) {
|
|
525
|
-
return assertBackendReport(summarizeBackendIntegrity(records), opts);
|
|
526
|
-
}
|
|
527
|
-
function assertBackendReport(report, opts) {
|
|
528
|
-
const allowMixed = opts.allowMixed ?? true;
|
|
529
|
-
if (report.verdict === "stub") throw new BackendIntegrityError(`backend-integrity: ran against a stub or unconfigured backend — ${report.diagnosis}`, report);
|
|
530
|
-
if (!allowMixed && report.verdict === "mixed") throw new BackendIntegrityError(`backend-integrity: partial backend failure rejected — ${report.diagnosis}`, report);
|
|
531
|
-
return report;
|
|
532
|
-
}
|
|
533
|
-
//#endregion
|
|
534
291
|
//#region src/campaign/judge-cell.ts
|
|
535
292
|
/**
|
|
536
293
|
* Judge scoring for one cell, with the paid-call accounting guard: a judge
|
|
@@ -1218,156 +975,6 @@ async function runEval(opts) {
|
|
|
1218
975
|
return runCampaign(opts);
|
|
1219
976
|
}
|
|
1220
977
|
//#endregion
|
|
1221
|
-
//#region src/campaign/surface-identity.ts
|
|
1222
|
-
const GIT_OBJECT_ID = /^(?:[a-f0-9]{40}|[a-f0-9]{64})$/;
|
|
1223
|
-
const SHA256 = /^sha256:[a-f0-9]{64}$/;
|
|
1224
|
-
/** Validate the immutable identity shape; the owning executor verifies the Git objects and patch. */
|
|
1225
|
-
function assertCodeSurfaceIdentity(surface) {
|
|
1226
|
-
if (!surface || typeof surface !== "object") throw new TypeError("CodeSurface must be an object");
|
|
1227
|
-
const candidate = surface;
|
|
1228
|
-
if (candidate.kind !== "code") throw new TypeError("CodeSurface.kind must be \"code\"");
|
|
1229
|
-
if (typeof candidate.worktreeRef !== "string" || candidate.worktreeRef.trim().length === 0) throw new TypeError("CodeSurface.worktreeRef must be a non-empty locator");
|
|
1230
|
-
if (typeof candidate.baseRef !== "string" || candidate.baseRef.trim().length === 0) throw new TypeError("CodeSurface.baseRef must be a non-empty ref label");
|
|
1231
|
-
for (const [field, value] of [
|
|
1232
|
-
["baseCommit", candidate.baseCommit],
|
|
1233
|
-
["baseTree", candidate.baseTree],
|
|
1234
|
-
["candidateCommit", candidate.candidateCommit],
|
|
1235
|
-
["candidateTree", candidate.candidateTree]
|
|
1236
|
-
]) if (typeof value !== "string" || !GIT_OBJECT_ID.test(value)) throw new TypeError(`CodeSurface.${field} must be a full Git object id`);
|
|
1237
|
-
const patch = candidate.patch;
|
|
1238
|
-
if (!patch || typeof patch !== "object" || patch.format !== "git-diff-binary") throw new TypeError("CodeSurface.patch.format must be \"git-diff-binary\"");
|
|
1239
|
-
if (typeof patch.sha256 !== "string" || !SHA256.test(patch.sha256)) throw new TypeError("CodeSurface.patch.sha256 must be a sha256 digest");
|
|
1240
|
-
if (!Number.isSafeInteger(patch.byteLength) || patch.byteLength < 0) throw new TypeError("CodeSurface.patch.byteLength must be a non-negative safe integer");
|
|
1241
|
-
}
|
|
1242
|
-
/** Assert that a value is a valid non-empty component surface. */
|
|
1243
|
-
function assertComponentSurface(surface) {
|
|
1244
|
-
if (!surface || typeof surface !== "object") throw new TypeError("ComponentSurface must be an object");
|
|
1245
|
-
const candidate = surface;
|
|
1246
|
-
if (candidate.kind !== "components") throw new TypeError("ComponentSurface.kind must be \"components\"");
|
|
1247
|
-
if (!candidate.components || typeof candidate.components !== "object" || Array.isArray(candidate.components)) throw new TypeError("ComponentSurface.components must be an object");
|
|
1248
|
-
const entries = Object.entries(candidate.components);
|
|
1249
|
-
if (entries.length === 0) throw new TypeError("ComponentSurface.components must not be empty");
|
|
1250
|
-
for (const [name, content] of entries) {
|
|
1251
|
-
if (!name.trim() || name.trim() !== name) throw new TypeError("ComponentSurface component names must be trimmed and non-empty");
|
|
1252
|
-
if (typeof content !== "string") throw new TypeError(`ComponentSurface component '${name}' must be a string`);
|
|
1253
|
-
}
|
|
1254
|
-
}
|
|
1255
|
-
/**
|
|
1256
|
-
* Deterministic identity material for a component surface.
|
|
1257
|
-
*
|
|
1258
|
-
* `canonicalString` orders keys by UTF-16 code unit (RFC 8785), which is a
|
|
1259
|
-
* property of the value alone. The previous material ordered them with
|
|
1260
|
-
* `localeCompare`, which reads the host's collation — so the same surface
|
|
1261
|
-
* could produce two different identities on two machines, and the stored
|
|
1262
|
-
* identity would stop matching a recomputation of the identical surface.
|
|
1263
|
-
*/
|
|
1264
|
-
function componentSurfaceIdentityMaterial(surface) {
|
|
1265
|
-
assertComponentSurface(surface);
|
|
1266
|
-
return canonicalString({
|
|
1267
|
-
schema: "tangle.component-surface",
|
|
1268
|
-
components: surface.components
|
|
1269
|
-
});
|
|
1270
|
-
}
|
|
1271
|
-
/**
|
|
1272
|
-
* The retired material builder, kept PRIVATE and reachable only from
|
|
1273
|
-
* {@link surfaceHashMatches}.
|
|
1274
|
-
*
|
|
1275
|
-
* A surface identity recorded before this release was minted from these bytes.
|
|
1276
|
-
* The verify path tries the current material first and falls back to this one,
|
|
1277
|
-
* so a stored identity still matches its own surface; nothing mints from it.
|
|
1278
|
-
*
|
|
1279
|
-
* Every component surface's identity moves, not only one whose names sort
|
|
1280
|
-
* differently under the host's collation: RFC 8785 also orders the two
|
|
1281
|
-
* top-level keys, so `components` precedes `schema` where this builder emitted
|
|
1282
|
-
* them in literal order. The retention window therefore covers every stored
|
|
1283
|
-
* component-surface identity, which is why this builder is kept rather than
|
|
1284
|
-
* scoped to the mixed-case case.
|
|
1285
|
-
*/
|
|
1286
|
-
function retiredComponentSurfaceIdentityMaterial(surface) {
|
|
1287
|
-
assertComponentSurface(surface);
|
|
1288
|
-
return JSON.stringify({
|
|
1289
|
-
schema: "tangle.component-surface",
|
|
1290
|
-
components: Object.fromEntries(Object.entries(surface.components).sort(([left], [right]) => left.localeCompare(right)))
|
|
1291
|
-
});
|
|
1292
|
-
}
|
|
1293
|
-
/** Canonical, location-independent identity of a finalized code candidate.
|
|
1294
|
-
* Commit metadata is excluded: two commits with the same base, final tree,
|
|
1295
|
-
* and patch bytes are the same executable candidate. */
|
|
1296
|
-
function codeSurfaceIdentityMaterial(surface) {
|
|
1297
|
-
assertCodeSurfaceIdentity(surface);
|
|
1298
|
-
return JSON.stringify({
|
|
1299
|
-
schema: "tangle.code-surface",
|
|
1300
|
-
baseCommit: surface.baseCommit,
|
|
1301
|
-
baseTree: surface.baseTree,
|
|
1302
|
-
candidateTree: surface.candidateTree,
|
|
1303
|
-
patch: {
|
|
1304
|
-
format: surface.patch.format,
|
|
1305
|
-
sha256: surface.patch.sha256,
|
|
1306
|
-
byteLength: surface.patch.byteLength
|
|
1307
|
-
}
|
|
1308
|
-
});
|
|
1309
|
-
}
|
|
1310
|
-
/** Full SHA-256 content identity for a prompt or finalized code surface. */
|
|
1311
|
-
function surfaceContentHash(surface) {
|
|
1312
|
-
const material = typeof surface === "string" ? surface : surface.kind === "components" ? componentSurfaceIdentityMaterial(surface) : codeSurfaceIdentityMaterial(surface);
|
|
1313
|
-
return `sha256:${createHash("sha256").update(material).digest("hex")}`;
|
|
1314
|
-
}
|
|
1315
|
-
/** Short loop key derived from the same content identity as provenance. */
|
|
1316
|
-
function surfaceHash(surface) {
|
|
1317
|
-
return surfaceContentHash(surface).slice(7, 23);
|
|
1318
|
-
}
|
|
1319
|
-
/**
|
|
1320
|
-
* Whether `storedHash` is the loop key of `surface`, under the current identity
|
|
1321
|
-
* material or the retired one.
|
|
1322
|
-
*
|
|
1323
|
-
* A stored key is 16 hex characters with no room for a scheme tag, so the
|
|
1324
|
-
* scheme cannot be read off the value the way an `agent-profile-cell` id names
|
|
1325
|
-
* its own. The verify path therefore tries both, which gives the same property:
|
|
1326
|
-
* a key minted by an earlier release still matches its own surface, and a
|
|
1327
|
-
* surface that was actually edited matches neither.
|
|
1328
|
-
*
|
|
1329
|
-
* Only a component surface can differ between the two; a prompt or code surface
|
|
1330
|
-
* produces identical material under both, so the second comparison is a no-op
|
|
1331
|
-
* for them.
|
|
1332
|
-
*/
|
|
1333
|
-
function surfaceHashMatches(surface, storedHash) {
|
|
1334
|
-
if (surfaceHash(surface) === storedHash) return true;
|
|
1335
|
-
if (typeof surface === "string" || surface.kind !== "components") return false;
|
|
1336
|
-
return createHash("sha256").update(retiredComponentSurfaceIdentityMaterial(surface)).digest("hex").slice(0, 16) === storedHash;
|
|
1337
|
-
}
|
|
1338
|
-
/** Canonical customer-visible description of the exact before/after surfaces. */
|
|
1339
|
-
function renderSurfaceDiff(winnerSurface, baselineSurface) {
|
|
1340
|
-
if (typeof winnerSurface === "string" && typeof baselineSurface === "string") return [
|
|
1341
|
-
"--- baseline",
|
|
1342
|
-
"+++ winner",
|
|
1343
|
-
...baselineSurface.split("\n").map((line) => `- ${line}`),
|
|
1344
|
-
...winnerSurface.split("\n").map((line) => `+ ${line}`)
|
|
1345
|
-
].join("\n");
|
|
1346
|
-
const describe = (surface) => {
|
|
1347
|
-
if (typeof surface === "string") return "(prompt surface)";
|
|
1348
|
-
if (surface.kind === "components") {
|
|
1349
|
-
assertComponentSurface(surface);
|
|
1350
|
-
return Object.entries(surface.components).sort(([left], [right]) => left.localeCompare(right)).map(([name, content]) => `[${name}]\n${content}`).join("\n\n");
|
|
1351
|
-
}
|
|
1352
|
-
assertCodeSurfaceIdentity(surface);
|
|
1353
|
-
return [
|
|
1354
|
-
`baseCommit=${surface.baseCommit}`,
|
|
1355
|
-
`baseTree=${surface.baseTree}`,
|
|
1356
|
-
`candidateCommit=${surface.candidateCommit}`,
|
|
1357
|
-
`candidateTree=${surface.candidateTree}`,
|
|
1358
|
-
`patch=${surface.patch.sha256}`,
|
|
1359
|
-
`patchBytes=${surface.patch.byteLength}`,
|
|
1360
|
-
...surface.summary ? [surface.summary] : []
|
|
1361
|
-
].join("\n");
|
|
1362
|
-
};
|
|
1363
|
-
return `--- baseline\n${describe(baselineSurface)}\n+++ winner\n${describe(winnerSurface)}`;
|
|
1364
|
-
}
|
|
1365
|
-
/** Bind a campaign cache entry to the exact surface and caller-owned execution revision. */
|
|
1366
|
-
function surfaceDispatchRef(surface, executionRef = "anonymous") {
|
|
1367
|
-
if (!executionRef.trim() || executionRef.trim() !== executionRef) throw new Error("surfaceDispatchRef: executionRef must be trimmed and non-empty");
|
|
1368
|
-
return `surface:${executionRef}:${surfaceContentHash(surface)}`;
|
|
1369
|
-
}
|
|
1370
|
-
//#endregion
|
|
1371
978
|
//#region src/canary.ts
|
|
1372
979
|
/**
|
|
1373
980
|
* Run all configured canaries against a chronological run list.
|
|
@@ -2792,403 +2399,767 @@ function paretoFrontierWithCrowding(candidates, objectives) {
|
|
|
2792
2399
|
return crowdingDistance(frontier, objectives).sort((a, b) => b.distance - a.distance);
|
|
2793
2400
|
}
|
|
2794
2401
|
//#endregion
|
|
2795
|
-
//#region src/
|
|
2402
|
+
//#region src/campaign/search-ledger.ts
|
|
2796
2403
|
/**
|
|
2797
|
-
*
|
|
2798
|
-
* out of a model response (reflective-mutation proposals, judge scores, the
|
|
2799
|
-
* completion-correctness checker).
|
|
2404
|
+
* Durable append-only audit log for improvement searches.
|
|
2800
2405
|
*
|
|
2801
|
-
*
|
|
2802
|
-
*
|
|
2803
|
-
*
|
|
2804
|
-
*
|
|
2805
|
-
*
|
|
2806
|
-
*
|
|
2807
|
-
*
|
|
2808
|
-
|
|
2809
|
-
|
|
2810
|
-
*
|
|
2811
|
-
*
|
|
2812
|
-
* balanced returns it unchanged. If a string was open at end-of-input we
|
|
2813
|
-
* also close it with `"` first, since a truncated string-mid-value is the
|
|
2814
|
-
* most common LLM cap-hit failure mode and JSON.parse cannot proceed
|
|
2815
|
-
* without one.
|
|
2406
|
+
* Existing campaign artifacts keep their own rich records: `RunRecord` owns a
|
|
2407
|
+
* measured run and `CostLedger` owns per-call accounting. This ledger does not
|
|
2408
|
+
* copy those structures. It binds their immutable ids and receipts into one replayable event stream so a
|
|
2409
|
+
* search can answer, after a crash, exactly which candidates and task attempts
|
|
2410
|
+
* existed, which surfaces actually fired, what they cost, and why they were
|
|
2411
|
+
* selected or rejected.
|
|
2412
|
+
*
|
|
2413
|
+
* The file format is canonical JSONL with a SHA-256 hash chain. Every append is
|
|
2414
|
+
* serialized across processes, fsynced before acknowledgement, and idempotent
|
|
2415
|
+
* by `eventId`. A malformed, non-canonical, truncated, reordered, or conflicting
|
|
2416
|
+
* log fails loudly; the implementation never skips a bad row.
|
|
2816
2417
|
*
|
|
2817
|
-
*
|
|
2818
|
-
*
|
|
2418
|
+
* The journal machinery itself (hash chain, locking, fsync, idempotent append)
|
|
2419
|
+
* is the generic `ledger-core` journal; this module supplies the campaign
|
|
2420
|
+
* codec: event schemas, canonical event ordering, and the search state machine.
|
|
2819
2421
|
*/
|
|
2820
|
-
|
|
2821
|
-
|
|
2822
|
-
|
|
2823
|
-
|
|
2824
|
-
|
|
2825
|
-
|
|
2826
|
-
|
|
2827
|
-
|
|
2422
|
+
const SEARCH_LEDGER_SCHEMA = "tangle.search-ledger.v1";
|
|
2423
|
+
const NON_EMPTY = z.string().min(1).refine((value) => value.trim() === value, "must not contain surrounding whitespace");
|
|
2424
|
+
const HASH = z.string().regex(/^sha256:[a-f0-9]{64}$/);
|
|
2425
|
+
const LINEAGE_NODE_ID = z.string().regex(/^[a-f0-9]{16}$/);
|
|
2426
|
+
const IMMUTABLE_REVISION = z.string().regex(/^(?:[a-f0-9]{40}|[a-f0-9]{64}|sha256:[a-f0-9]{64}|sha512:[A-Za-z0-9+/=]+)$/);
|
|
2427
|
+
const ISO_TIMESTAMP = z.string().regex(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/).refine((value) => Number.isFinite(Date.parse(value)), "invalid timestamp");
|
|
2428
|
+
const NON_NEGATIVE_INT = z.number().int().nonnegative().safe();
|
|
2429
|
+
const FINITE_NUMBER = z.number().finite();
|
|
2430
|
+
const ArtifactRefSchema = z.object({
|
|
2431
|
+
role: NON_EMPTY,
|
|
2432
|
+
uri: NON_EMPTY,
|
|
2433
|
+
sha256: HASH,
|
|
2434
|
+
byteLength: NON_NEGATIVE_INT
|
|
2435
|
+
}).strict();
|
|
2436
|
+
const SourceRefSchema = z.object({
|
|
2437
|
+
uri: NON_EMPTY,
|
|
2438
|
+
revision: IMMUTABLE_REVISION
|
|
2439
|
+
}).strict();
|
|
2440
|
+
const FailureReasonSchema = z.object({
|
|
2441
|
+
code: NON_EMPTY,
|
|
2442
|
+
message: NON_EMPTY
|
|
2443
|
+
}).strict();
|
|
2444
|
+
const EventBaseShape = {
|
|
2445
|
+
eventId: NON_EMPTY,
|
|
2446
|
+
occurredAt: ISO_TIMESTAMP,
|
|
2447
|
+
artifacts: z.array(ArtifactRefSchema).min(1)
|
|
2448
|
+
};
|
|
2449
|
+
const OperationKindSchema = z.enum([
|
|
2450
|
+
"candidate-generation",
|
|
2451
|
+
"analysis",
|
|
2452
|
+
"selection",
|
|
2453
|
+
"judge",
|
|
2454
|
+
"other"
|
|
2455
|
+
]);
|
|
2456
|
+
const CandidateSlotSchema = z.object({
|
|
2457
|
+
slotId: NON_EMPTY,
|
|
2458
|
+
generationOperationId: NON_EMPTY
|
|
2459
|
+
}).strict();
|
|
2460
|
+
const PlannedOperationSchema = z.object({
|
|
2461
|
+
operationId: NON_EMPTY,
|
|
2462
|
+
kind: OperationKindSchema
|
|
2463
|
+
}).strict();
|
|
2464
|
+
const SearchPlanExtendedSchema = z.object({
|
|
2465
|
+
...EventBaseShape,
|
|
2466
|
+
kind: z.literal("search-plan-extended"),
|
|
2467
|
+
extension: z.object({
|
|
2468
|
+
candidateSlots: z.array(CandidateSlotSchema),
|
|
2469
|
+
operations: z.array(PlannedOperationSchema)
|
|
2470
|
+
}).strict().superRefine((extension, ctx) => {
|
|
2471
|
+
if (extension.candidateSlots.length === 0 && extension.operations.length === 0) ctx.addIssue({
|
|
2472
|
+
code: "custom",
|
|
2473
|
+
message: "a plan extension must add slots or operations"
|
|
2474
|
+
});
|
|
2475
|
+
})
|
|
2476
|
+
}).strict();
|
|
2477
|
+
const SearchPlannedSchema = z.object({
|
|
2478
|
+
...EventBaseShape,
|
|
2479
|
+
kind: z.literal("search-planned"),
|
|
2480
|
+
plan: z.object({
|
|
2481
|
+
candidateSlots: z.array(CandidateSlotSchema).min(1),
|
|
2482
|
+
tasks: z.array(z.object({
|
|
2483
|
+
taskId: NON_EMPTY,
|
|
2484
|
+
source: SourceRefSchema,
|
|
2485
|
+
benchmark: SourceRefSchema,
|
|
2486
|
+
maxAttempts: z.number().int().positive().safe()
|
|
2487
|
+
}).strict()).min(1),
|
|
2488
|
+
operations: z.array(PlannedOperationSchema).min(1)
|
|
2489
|
+
}).strict()
|
|
2490
|
+
}).strict();
|
|
2491
|
+
const CandidateRegisteredSchema = z.object({
|
|
2492
|
+
...EventBaseShape,
|
|
2493
|
+
kind: z.literal("candidate-registered"),
|
|
2494
|
+
slotId: NON_EMPTY,
|
|
2495
|
+
generationOperationId: NON_EMPTY,
|
|
2496
|
+
candidateId: NON_EMPTY,
|
|
2497
|
+
lineage: z.object({
|
|
2498
|
+
lineageNodeId: LINEAGE_NODE_ID,
|
|
2499
|
+
parentCandidateIds: z.array(NON_EMPTY),
|
|
2500
|
+
generation: NON_NEGATIVE_INT,
|
|
2501
|
+
proposer: NON_EMPTY,
|
|
2502
|
+
proposerSource: SourceRefSchema
|
|
2503
|
+
}).strict(),
|
|
2504
|
+
surfaces: z.array(z.object({
|
|
2505
|
+
surfaceId: NON_EMPTY,
|
|
2506
|
+
kind: z.enum([
|
|
2507
|
+
"prompt",
|
|
2508
|
+
"tool-contract",
|
|
2509
|
+
"runtime-config",
|
|
2510
|
+
"memory",
|
|
2511
|
+
"knowledge",
|
|
2512
|
+
"agent-profile",
|
|
2513
|
+
"code",
|
|
2514
|
+
"deployment"
|
|
2515
|
+
]),
|
|
2516
|
+
artifact: ArtifactRefSchema
|
|
2517
|
+
}).strict()).min(1)
|
|
2518
|
+
}).strict();
|
|
2519
|
+
const CandidateSlotClosedSchema = z.object({
|
|
2520
|
+
...EventBaseShape,
|
|
2521
|
+
kind: z.literal("candidate-slot-closed"),
|
|
2522
|
+
slotId: NON_EMPTY,
|
|
2523
|
+
generationOperationId: NON_EMPTY,
|
|
2524
|
+
reason: FailureReasonSchema
|
|
2525
|
+
}).strict();
|
|
2526
|
+
const KnownTokensSchema = z.object({
|
|
2527
|
+
status: z.literal("known"),
|
|
2528
|
+
inputTokens: NON_NEGATIVE_INT,
|
|
2529
|
+
outputTokens: NON_NEGATIVE_INT,
|
|
2530
|
+
cachedTokens: NON_NEGATIVE_INT
|
|
2531
|
+
}).strict();
|
|
2532
|
+
const UnknownSchema = z.object({
|
|
2533
|
+
status: z.literal("unknown"),
|
|
2534
|
+
reason: NON_EMPTY
|
|
2535
|
+
}).strict();
|
|
2536
|
+
const KnownCostSchema = z.object({
|
|
2537
|
+
status: z.literal("known"),
|
|
2538
|
+
usd: z.number().finite().nonnegative(),
|
|
2539
|
+
source: z.enum([
|
|
2540
|
+
"provider",
|
|
2541
|
+
"pricing-table",
|
|
2542
|
+
"free"
|
|
2543
|
+
])
|
|
2544
|
+
}).strict().superRefine((cost, ctx) => {
|
|
2545
|
+
if (cost.source === "free" && cost.usd !== 0) ctx.addIssue({
|
|
2546
|
+
code: "custom",
|
|
2547
|
+
message: "free cost source must have usd 0"
|
|
2548
|
+
});
|
|
2549
|
+
});
|
|
2550
|
+
const UnknownCostSchema = z.object({
|
|
2551
|
+
status: z.literal("unknown"),
|
|
2552
|
+
knownLowerBoundUsd: z.number().finite().nonnegative(),
|
|
2553
|
+
reason: NON_EMPTY
|
|
2554
|
+
}).strict();
|
|
2555
|
+
const AccountingSchema = z.object({
|
|
2556
|
+
tokens: z.discriminatedUnion("status", [KnownTokensSchema, UnknownSchema]),
|
|
2557
|
+
cost: z.discriminatedUnion("status", [KnownCostSchema, UnknownCostSchema])
|
|
2558
|
+
}).strict();
|
|
2559
|
+
const MetricsSchema = z.record(NON_EMPTY, FINITE_NUMBER).superRefine((metrics, ctx) => {
|
|
2560
|
+
for (const key of Object.keys(metrics)) if (key === "__proto__" || key === "constructor" || key === "prototype") ctx.addIssue({
|
|
2561
|
+
code: "custom",
|
|
2562
|
+
message: `unsafe metric key ${key}`
|
|
2563
|
+
});
|
|
2564
|
+
});
|
|
2565
|
+
const OutcomeSchema = z.discriminatedUnion("status", [
|
|
2566
|
+
z.object({
|
|
2567
|
+
status: z.literal("passed"),
|
|
2568
|
+
score: FINITE_NUMBER,
|
|
2569
|
+
metrics: MetricsSchema
|
|
2570
|
+
}).strict(),
|
|
2571
|
+
z.object({
|
|
2572
|
+
status: z.literal("failed"),
|
|
2573
|
+
score: FINITE_NUMBER,
|
|
2574
|
+
metrics: MetricsSchema,
|
|
2575
|
+
failure: FailureReasonSchema
|
|
2576
|
+
}).strict(),
|
|
2577
|
+
z.object({
|
|
2578
|
+
status: z.literal("errored"),
|
|
2579
|
+
metrics: MetricsSchema,
|
|
2580
|
+
error: FailureReasonSchema.extend({ retryable: z.boolean() }).strict()
|
|
2581
|
+
}).strict()
|
|
2582
|
+
]);
|
|
2583
|
+
const EffectSchema = z.discriminatedUnion("status", [z.object({
|
|
2584
|
+
status: z.literal("measured"),
|
|
2585
|
+
metric: NON_EMPTY,
|
|
2586
|
+
baselineValue: FINITE_NUMBER,
|
|
2587
|
+
candidateValue: FINITE_NUMBER,
|
|
2588
|
+
delta: FINITE_NUMBER
|
|
2589
|
+
}).strict().superRefine((effect, ctx) => {
|
|
2590
|
+
const expected = effect.candidateValue - effect.baselineValue;
|
|
2591
|
+
const tolerance = Number.EPSILON * Math.max(1, Math.abs(expected), Math.abs(effect.delta)) * 8;
|
|
2592
|
+
if (Math.abs(effect.delta - expected) > tolerance) ctx.addIssue({
|
|
2593
|
+
code: "custom",
|
|
2594
|
+
message: "delta must equal candidateValue - baselineValue"
|
|
2595
|
+
});
|
|
2596
|
+
}), z.object({
|
|
2597
|
+
status: z.literal("not-measured"),
|
|
2598
|
+
reason: NON_EMPTY
|
|
2599
|
+
}).strict()]);
|
|
2600
|
+
const SurfaceEvidenceSchema = z.object({
|
|
2601
|
+
surfaceId: NON_EMPTY,
|
|
2602
|
+
fired: z.boolean(),
|
|
2603
|
+
firingCount: NON_NEGATIVE_INT,
|
|
2604
|
+
effect: EffectSchema,
|
|
2605
|
+
evidence: z.array(ArtifactRefSchema).min(1)
|
|
2606
|
+
}).strict().superRefine((evidence, ctx) => {
|
|
2607
|
+
if (evidence.fired && evidence.firingCount === 0) ctx.addIssue({
|
|
2608
|
+
code: "custom",
|
|
2609
|
+
message: "a fired surface must have firingCount >= 1"
|
|
2610
|
+
});
|
|
2611
|
+
if (!evidence.fired && evidence.firingCount !== 0) ctx.addIssue({
|
|
2612
|
+
code: "custom",
|
|
2613
|
+
message: "a surface that did not fire must have firingCount 0"
|
|
2614
|
+
});
|
|
2615
|
+
if (!evidence.fired && evidence.effect.status === "measured" && evidence.effect.delta !== 0) ctx.addIssue({
|
|
2616
|
+
code: "custom",
|
|
2617
|
+
message: "a surface that did not fire cannot claim non-zero effect"
|
|
2618
|
+
});
|
|
2619
|
+
});
|
|
2620
|
+
const TaskAttemptedSchema = z.object({
|
|
2621
|
+
...EventBaseShape,
|
|
2622
|
+
kind: z.literal("task-attempted"),
|
|
2623
|
+
candidateId: NON_EMPTY,
|
|
2624
|
+
runId: NON_EMPTY,
|
|
2625
|
+
attemptIndex: NON_NEGATIVE_INT,
|
|
2626
|
+
task: z.object({
|
|
2627
|
+
taskId: NON_EMPTY,
|
|
2628
|
+
source: SourceRefSchema
|
|
2629
|
+
}).strict(),
|
|
2630
|
+
identity: z.object({
|
|
2631
|
+
model: z.object({
|
|
2632
|
+
provider: NON_EMPTY,
|
|
2633
|
+
snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
|
|
2634
|
+
}).strict(),
|
|
2635
|
+
agent: SourceRefSchema,
|
|
2636
|
+
benchmark: SourceRefSchema
|
|
2637
|
+
}).strict(),
|
|
2638
|
+
outcome: OutcomeSchema,
|
|
2639
|
+
accounting: AccountingSchema,
|
|
2640
|
+
surfaceEvidence: z.array(SurfaceEvidenceSchema).min(1)
|
|
2641
|
+
}).strict();
|
|
2642
|
+
const SearchOperationRecordedSchema = z.object({
|
|
2643
|
+
...EventBaseShape,
|
|
2644
|
+
kind: z.literal("search-operation-recorded"),
|
|
2645
|
+
operationId: NON_EMPTY,
|
|
2646
|
+
operationKind: OperationKindSchema,
|
|
2647
|
+
execution: z.discriminatedUnion("kind", [z.object({
|
|
2648
|
+
kind: z.literal("model"),
|
|
2649
|
+
model: z.object({
|
|
2650
|
+
provider: NON_EMPTY,
|
|
2651
|
+
snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
|
|
2652
|
+
}).strict(),
|
|
2653
|
+
source: SourceRefSchema
|
|
2654
|
+
}).strict(), z.object({
|
|
2655
|
+
kind: z.literal("deterministic"),
|
|
2656
|
+
source: SourceRefSchema
|
|
2657
|
+
}).strict()]),
|
|
2658
|
+
outcome: z.discriminatedUnion("status", [
|
|
2659
|
+
z.object({ status: z.literal("completed") }).strict(),
|
|
2660
|
+
z.object({
|
|
2661
|
+
status: z.literal("partial"),
|
|
2662
|
+
failure: FailureReasonSchema
|
|
2663
|
+
}).strict(),
|
|
2664
|
+
z.object({
|
|
2665
|
+
status: z.literal("failed"),
|
|
2666
|
+
failure: FailureReasonSchema
|
|
2667
|
+
}).strict()
|
|
2668
|
+
]),
|
|
2669
|
+
accounting: AccountingSchema
|
|
2670
|
+
}).strict();
|
|
2671
|
+
const CandidateDecidedSchema = z.object({
|
|
2672
|
+
...EventBaseShape,
|
|
2673
|
+
kind: z.literal("candidate-decided"),
|
|
2674
|
+
candidateId: NON_EMPTY,
|
|
2675
|
+
decision: z.discriminatedUnion("status", [z.object({ status: z.literal("selected") }).strict(), z.object({
|
|
2676
|
+
status: z.literal("rejected"),
|
|
2677
|
+
reason: FailureReasonSchema
|
|
2678
|
+
}).strict()])
|
|
2679
|
+
}).strict();
|
|
2680
|
+
const SearchCompletedSchema = z.object({
|
|
2681
|
+
...EventBaseShape,
|
|
2682
|
+
kind: z.literal("search-completed"),
|
|
2683
|
+
result: z.discriminatedUnion("status", [z.object({
|
|
2684
|
+
status: z.literal("selected"),
|
|
2685
|
+
candidateId: NON_EMPTY
|
|
2686
|
+
}).strict(), z.object({
|
|
2687
|
+
status: z.literal("all-rejected"),
|
|
2688
|
+
reason: FailureReasonSchema
|
|
2689
|
+
}).strict()])
|
|
2690
|
+
}).strict();
|
|
2691
|
+
const EventSchema = z.discriminatedUnion("kind", [
|
|
2692
|
+
SearchPlannedSchema,
|
|
2693
|
+
SearchPlanExtendedSchema,
|
|
2694
|
+
CandidateRegisteredSchema,
|
|
2695
|
+
CandidateSlotClosedSchema,
|
|
2696
|
+
TaskAttemptedSchema,
|
|
2697
|
+
SearchOperationRecordedSchema,
|
|
2698
|
+
CandidateDecidedSchema,
|
|
2699
|
+
SearchCompletedSchema
|
|
2700
|
+
]);
|
|
2701
|
+
const EntrySchema = z.object({
|
|
2702
|
+
schema: z.literal(SEARCH_LEDGER_SCHEMA),
|
|
2703
|
+
campaignId: NON_EMPTY,
|
|
2704
|
+
sequence: NON_NEGATIVE_INT,
|
|
2705
|
+
previousHash: z.union([HASH, z.null()]),
|
|
2706
|
+
event: EventSchema,
|
|
2707
|
+
entryHash: HASH
|
|
2708
|
+
}).strict();
|
|
2709
|
+
/** Validate and return a canonical copy. Arrays whose order is not semantic are
|
|
2710
|
+
* sorted so retries from different processes produce byte-identical events. */
|
|
2711
|
+
function validateSearchLedgerEvent(input) {
|
|
2712
|
+
const parsed = EventSchema.safeParse(input);
|
|
2713
|
+
if (!parsed.success) throw new SearchLedgerError(`invalid search ledger event: ${formatZodError(parsed.error)}`);
|
|
2714
|
+
return normalizeEvent(parsed.data);
|
|
2715
|
+
}
|
|
2716
|
+
/** Open a durable filesystem search ledger. Construction performs no I/O; the
|
|
2717
|
+
* first `append` or `replay` validates the complete existing file. */
|
|
2718
|
+
function openSearchLedger(options) {
|
|
2719
|
+
if (options.path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
|
|
2720
|
+
return new FileSearchLedger(options.path, options.campaignId, options.trustedHead);
|
|
2721
|
+
}
|
|
2722
|
+
/** Replay immutable search-ledger JSONL through the same codec as FileSearchLedger. */
|
|
2723
|
+
function replaySearchLedgerText(text, campaignId, source) {
|
|
2724
|
+
return replayLedgerText(text, source, searchLedgerCodec(campaignId)).projection;
|
|
2725
|
+
}
|
|
2726
|
+
function searchLedgerCodec(campaignId) {
|
|
2727
|
+
return {
|
|
2728
|
+
...SEARCH_LEDGER_FILE_CONTEXT,
|
|
2729
|
+
header: {
|
|
2730
|
+
schema: SEARCH_LEDGER_SCHEMA,
|
|
2731
|
+
campaignId
|
|
2732
|
+
},
|
|
2733
|
+
conflictError: (message) => new SearchLedgerConflictError(message),
|
|
2734
|
+
parseEntry: parseSearchLedgerEntry,
|
|
2735
|
+
checkEntryHeader: (entry, index) => {
|
|
2736
|
+
if (entry.campaignId !== campaignId) throw new SearchLedgerIntegrityError(`entry ${index} belongs to campaign ${entry.campaignId}, expected ${campaignId}`);
|
|
2737
|
+
},
|
|
2738
|
+
createProjector: () => createSearchLedgerProjector(campaignId)
|
|
2739
|
+
};
|
|
2740
|
+
}
|
|
2741
|
+
/** Append-only file-backed search ledger with idempotent writes and replay. */
|
|
2742
|
+
var FileSearchLedger = class {
|
|
2743
|
+
path;
|
|
2744
|
+
campaignId;
|
|
2745
|
+
trustedHeadPath;
|
|
2746
|
+
trustedHeadMode;
|
|
2747
|
+
journal;
|
|
2748
|
+
constructor(path, campaignId, trustedHead = "pin") {
|
|
2749
|
+
if (path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
|
|
2750
|
+
if (campaignId.length === 0) throw new SearchLedgerError("campaignId is empty");
|
|
2751
|
+
if (campaignId.trim() !== campaignId) throw new SearchLedgerError("campaignId must not contain surrounding whitespace");
|
|
2752
|
+
this.campaignId = campaignId;
|
|
2753
|
+
this.trustedHeadMode = trustedHead;
|
|
2754
|
+
this.journal = new FileLedgerJournal(path, searchLedgerCodec(campaignId), { requireTrustedHead: trustedHead === "require" });
|
|
2755
|
+
this.path = this.journal.path;
|
|
2756
|
+
this.trustedHeadPath = this.journal.trustedHeadPath;
|
|
2757
|
+
}
|
|
2758
|
+
async replay() {
|
|
2759
|
+
return (await this.journal.replay()).projection;
|
|
2760
|
+
}
|
|
2761
|
+
async append(input) {
|
|
2762
|
+
const event = validateSearchLedgerEvent(input);
|
|
2763
|
+
const { entry, appended, projection } = await this.journal.append(event, { pinHead: this.trustedHeadMode !== "off" });
|
|
2764
|
+
return {
|
|
2765
|
+
entry,
|
|
2766
|
+
appended,
|
|
2767
|
+
replay: projection
|
|
2768
|
+
};
|
|
2769
|
+
}
|
|
2770
|
+
async trustedHead() {
|
|
2771
|
+
return this.journal.trustedHead();
|
|
2772
|
+
}
|
|
2773
|
+
async pinTrustedHead() {
|
|
2774
|
+
return this.journal.pinTrustedHead();
|
|
2775
|
+
}
|
|
2776
|
+
async clearTrustedHead() {
|
|
2777
|
+
return this.journal.clearTrustedHead();
|
|
2778
|
+
}
|
|
2779
|
+
};
|
|
2780
|
+
function parseSearchLedgerEntry(raw, context) {
|
|
2781
|
+
const parsed = EntrySchema.safeParse(raw);
|
|
2782
|
+
if (!parsed.success) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has a malformed entry at line ${context.line}: ${formatZodError(parsed.error)}`);
|
|
2783
|
+
const entry = parsed.data;
|
|
2784
|
+
if (canonicalString(validateSearchLedgerEvent(entry.event)) !== canonicalString(entry.event)) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has non-canonical event ordering at line ${context.line}`);
|
|
2785
|
+
return entry;
|
|
2786
|
+
}
|
|
2787
|
+
/** Replay the campaign search state machine over chain-verified entries. The
|
|
2788
|
+
* generic journal owns sequence, hash, and eventId-uniqueness checks; this
|
|
2789
|
+
* projector owns every campaign invariant and builds the replay projection. */
|
|
2790
|
+
function createSearchLedgerProjector(campaignId) {
|
|
2791
|
+
const candidates = /* @__PURE__ */ new Map();
|
|
2792
|
+
const plannedSlots = /* @__PURE__ */ new Map();
|
|
2793
|
+
const plannedOperations = /* @__PURE__ */ new Map();
|
|
2794
|
+
const planExtensions = [];
|
|
2795
|
+
const candidateBySlot = /* @__PURE__ */ new Map();
|
|
2796
|
+
const closedSlots = /* @__PURE__ */ new Map();
|
|
2797
|
+
const lineageNodes = /* @__PURE__ */ new Map();
|
|
2798
|
+
const runIds = /* @__PURE__ */ new Set();
|
|
2799
|
+
const attemptKeys = /* @__PURE__ */ new Set();
|
|
2800
|
+
const candidateEvents = [];
|
|
2801
|
+
const closedSlotEvents = [];
|
|
2802
|
+
const attempts = [];
|
|
2803
|
+
const operationEvents = [];
|
|
2804
|
+
const operationsById = /* @__PURE__ */ new Map();
|
|
2805
|
+
const decisions = [];
|
|
2806
|
+
let planEvent = null;
|
|
2807
|
+
let completion = null;
|
|
2808
|
+
let previousOccurredAt = Number.NEGATIVE_INFINITY;
|
|
2809
|
+
const apply = (entry, index) => {
|
|
2810
|
+
const event = entry.event;
|
|
2811
|
+
if (completion) throw new SearchLedgerIntegrityError(`event ${event.eventId} appears after terminal event ${completion.eventId}`);
|
|
2812
|
+
const occurredAt = Date.parse(event.occurredAt);
|
|
2813
|
+
if (occurredAt < previousOccurredAt) throw new SearchLedgerIntegrityError(`event ${event.eventId} occurred before the preceding durable event`);
|
|
2814
|
+
previousOccurredAt = occurredAt;
|
|
2815
|
+
assertUnique(event.artifacts.map(artifactKey), "artifact receipt", event.eventId);
|
|
2816
|
+
if (event.kind === "search-planned") {
|
|
2817
|
+
if (index !== 0 || planEvent) throw new SearchLedgerIntegrityError("search plan must be the first and only plan event");
|
|
2818
|
+
assertUnique(event.plan.candidateSlots.map((slot) => slot.slotId), "candidate slot", event.eventId);
|
|
2819
|
+
assertUnique(event.plan.tasks.map((task) => task.taskId), "planned taskId", event.eventId);
|
|
2820
|
+
assertUnique(event.plan.operations.map((operation) => operation.operationId), "planned operationId", event.eventId);
|
|
2821
|
+
for (const operation of event.plan.operations) plannedOperations.set(operation.operationId, operation);
|
|
2822
|
+
for (const slot of event.plan.candidateSlots) {
|
|
2823
|
+
assertSlotGenerationOperation(slot, plannedOperations);
|
|
2824
|
+
plannedSlots.set(slot.slotId, slot);
|
|
2825
|
+
}
|
|
2826
|
+
planEvent = event;
|
|
2827
|
+
return;
|
|
2828
2828
|
}
|
|
2829
|
-
if (
|
|
2830
|
-
|
|
2831
|
-
|
|
2832
|
-
|
|
2829
|
+
if (!planEvent) throw new SearchLedgerIntegrityError(`event ${event.eventId} appears before the required search plan`);
|
|
2830
|
+
if (event.kind === "search-plan-extended") {
|
|
2831
|
+
assertUnique(event.extension.candidateSlots.map((slot) => slot.slotId), "candidate slot", event.eventId);
|
|
2832
|
+
assertUnique(event.extension.operations.map((operation) => operation.operationId), "planned operationId", event.eventId);
|
|
2833
|
+
for (const operation of event.extension.operations) {
|
|
2834
|
+
if (plannedOperations.has(operation.operationId)) throw new SearchLedgerIntegrityError(`plan extension ${event.eventId} re-plans operation ${operation.operationId}`);
|
|
2835
|
+
plannedOperations.set(operation.operationId, operation);
|
|
2833
2836
|
}
|
|
2834
|
-
|
|
2835
|
-
|
|
2836
|
-
|
|
2837
|
+
for (const slot of event.extension.candidateSlots) {
|
|
2838
|
+
if (plannedSlots.has(slot.slotId)) throw new SearchLedgerIntegrityError(`plan extension ${event.eventId} re-plans candidate slot ${slot.slotId}`);
|
|
2839
|
+
assertSlotGenerationOperation(slot, plannedOperations);
|
|
2840
|
+
plannedSlots.set(slot.slotId, slot);
|
|
2837
2841
|
}
|
|
2838
|
-
|
|
2839
|
-
|
|
2840
|
-
if (c === "\"") {
|
|
2841
|
-
inString = true;
|
|
2842
|
-
continue;
|
|
2842
|
+
planExtensions.push(event);
|
|
2843
|
+
return;
|
|
2843
2844
|
}
|
|
2844
|
-
if (
|
|
2845
|
-
|
|
2846
|
-
|
|
2847
|
-
|
|
2848
|
-
if (
|
|
2845
|
+
if (event.kind === "candidate-registered") {
|
|
2846
|
+
if (candidates.has(event.candidateId)) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} was registered twice`);
|
|
2847
|
+
const plannedSlot = plannedSlots.get(event.slotId);
|
|
2848
|
+
if (!plannedSlot) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} binds unknown slot ${event.slotId}`);
|
|
2849
|
+
if (candidateBySlot.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was bound twice`);
|
|
2850
|
+
if (closedSlots.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was already closed`);
|
|
2851
|
+
if (event.generationOperationId !== plannedSlot.generationOperationId) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} generation operation ${event.generationOperationId} does not match slot ${event.slotId} plan ${plannedSlot.generationOperationId}`);
|
|
2852
|
+
const generationOperation = operationsById.get(event.generationOperationId);
|
|
2853
|
+
if (!generationOperation) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} precedes generation operation ${event.generationOperationId}`);
|
|
2854
|
+
if (generationOperation.outcome.status === "failed") throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} cannot bind failed generation operation ${event.generationOperationId}`);
|
|
2855
|
+
const previousCandidate = lineageNodes.get(event.lineage.lineageNodeId);
|
|
2856
|
+
if (previousCandidate) throw new SearchLedgerIntegrityError(`lineage node ${event.lineage.lineageNodeId} is already bound to ${previousCandidate}`);
|
|
2857
|
+
assertUnique(event.lineage.parentCandidateIds, "parentCandidateId", event.eventId);
|
|
2858
|
+
const parents = event.lineage.parentCandidateIds.map((id) => {
|
|
2859
|
+
const parent = candidates.get(id);
|
|
2860
|
+
if (!parent) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} references unknown parent ${id}`);
|
|
2861
|
+
return parent;
|
|
2862
|
+
});
|
|
2863
|
+
const expectedGeneration = parents.length === 0 ? 0 : Math.max(...parents.map((parent) => parent.registered.lineage.generation)) + 1;
|
|
2864
|
+
if (event.lineage.generation !== expectedGeneration) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} generation ${event.lineage.generation} does not follow its parents (expected ${expectedGeneration})`);
|
|
2865
|
+
assertUnique(event.surfaces.map((surface) => surface.surfaceId), "surfaceId", event.eventId);
|
|
2866
|
+
candidates.set(event.candidateId, {
|
|
2867
|
+
registered: event,
|
|
2868
|
+
attempts: [],
|
|
2869
|
+
decision: null
|
|
2870
|
+
});
|
|
2871
|
+
candidateBySlot.set(event.slotId, event.candidateId);
|
|
2872
|
+
lineageNodes.set(event.lineage.lineageNodeId, event.candidateId);
|
|
2873
|
+
candidateEvents.push(event);
|
|
2874
|
+
return;
|
|
2849
2875
|
}
|
|
2850
|
-
|
|
2851
|
-
|
|
2852
|
-
|
|
2853
|
-
|
|
2854
|
-
|
|
2855
|
-
|
|
2856
|
-
|
|
2857
|
-
|
|
2858
|
-
|
|
2859
|
-
|
|
2860
|
-
|
|
2861
|
-
|
|
2862
|
-
|
|
2863
|
-
|
|
2864
|
-
|
|
2865
|
-
|
|
2866
|
-
|
|
2867
|
-
|
|
2868
|
-
|
|
2869
|
-
|
|
2870
|
-
|
|
2871
|
-
|
|
2872
|
-
|
|
2873
|
-
|
|
2874
|
-
|
|
2875
|
-
|
|
2876
|
-
|
|
2877
|
-
|
|
2878
|
-
|
|
2879
|
-
|
|
2880
|
-
|
|
2876
|
+
if (event.kind === "task-attempted") {
|
|
2877
|
+
const candidate = candidates.get(event.candidateId);
|
|
2878
|
+
if (!candidate) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} references unknown candidate ${event.candidateId}`);
|
|
2879
|
+
if (candidate.decision) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} appears after candidate ${event.candidateId} was decided`);
|
|
2880
|
+
const plannedTask = planEvent.plan.tasks.find((task) => task.taskId === event.task.taskId);
|
|
2881
|
+
if (!plannedTask) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} references unplanned task ${event.task.taskId}`);
|
|
2882
|
+
if (canonicalString(plannedTask.source) !== canonicalString(event.task.source) || canonicalString(plannedTask.benchmark) !== canonicalString(event.identity.benchmark)) throw new SearchLedgerIntegrityError(`task ${event.task.taskId} does not match its planned source identity`);
|
|
2883
|
+
if (event.attemptIndex >= plannedTask.maxAttempts) throw new SearchLedgerIntegrityError(`task ${event.task.taskId} attempt ${event.attemptIndex} exceeds planned maxAttempts ${plannedTask.maxAttempts}`);
|
|
2884
|
+
if (runIds.has(event.runId)) throw new SearchLedgerIntegrityError(`runId ${event.runId} was recorded twice`);
|
|
2885
|
+
runIds.add(event.runId);
|
|
2886
|
+
const attemptKey = canonicalString([
|
|
2887
|
+
event.candidateId,
|
|
2888
|
+
event.task.taskId,
|
|
2889
|
+
event.attemptIndex
|
|
2890
|
+
]);
|
|
2891
|
+
if (attemptKeys.has(attemptKey)) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} attempt ${event.attemptIndex} was recorded twice`);
|
|
2892
|
+
const expectedAttemptIndex = candidate.attempts.filter((attempt) => attempt.task.taskId === event.task.taskId).length;
|
|
2893
|
+
if (event.attemptIndex !== expectedAttemptIndex) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} attempt index ${event.attemptIndex} is not contiguous (expected ${expectedAttemptIndex})`);
|
|
2894
|
+
const previousAttempt = candidate.attempts.find((attempt) => attempt.task.taskId === event.task.taskId);
|
|
2895
|
+
if (previousAttempt?.outcome.status !== void 0 && previousAttempt.outcome.status !== "errored") throw new SearchLedgerIntegrityError(`task ${event.task.taskId} was retried after a measured outcome`);
|
|
2896
|
+
if (previousAttempt && canonicalString({
|
|
2897
|
+
task: previousAttempt.task,
|
|
2898
|
+
identity: previousAttempt.identity
|
|
2899
|
+
}) !== canonicalString({
|
|
2900
|
+
task: event.task,
|
|
2901
|
+
identity: event.identity
|
|
2902
|
+
})) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} changed immutable execution identity between attempts`);
|
|
2903
|
+
attemptKeys.add(attemptKey);
|
|
2904
|
+
const declared = candidate.registered.surfaces.map((surface) => surface.surfaceId).sort();
|
|
2905
|
+
const observed = event.surfaceEvidence.map((surface) => surface.surfaceId).sort();
|
|
2906
|
+
assertUnique(observed, "surface evidence", event.eventId);
|
|
2907
|
+
for (const evidence of event.surfaceEvidence) assertUnique(evidence.evidence.map(artifactKey), "surface evidence receipt", event.eventId);
|
|
2908
|
+
if (canonicalString(declared) !== canonicalString(observed)) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} surface evidence does not exactly cover candidate ${event.candidateId}`);
|
|
2909
|
+
candidate.attempts.push(event);
|
|
2910
|
+
attempts.push(event);
|
|
2911
|
+
return;
|
|
2881
2912
|
}
|
|
2882
|
-
if (
|
|
2883
|
-
|
|
2884
|
-
|
|
2885
|
-
|
|
2913
|
+
if (event.kind === "search-operation-recorded") {
|
|
2914
|
+
const plannedOperation = plannedOperations.get(event.operationId);
|
|
2915
|
+
if (!plannedOperation) throw new SearchLedgerIntegrityError(`operation ${event.operationId} was not declared in the search plan`);
|
|
2916
|
+
if (plannedOperation.kind !== event.operationKind) throw new SearchLedgerIntegrityError(`operation ${event.operationId} kind ${event.operationKind} does not match planned ${plannedOperation.kind}`);
|
|
2917
|
+
if (operationsById.has(event.operationId)) throw new SearchLedgerIntegrityError(`operation ${event.operationId} was recorded twice`);
|
|
2918
|
+
operationsById.set(event.operationId, event);
|
|
2919
|
+
operationEvents.push(event);
|
|
2920
|
+
return;
|
|
2886
2921
|
}
|
|
2887
|
-
if (
|
|
2888
|
-
|
|
2889
|
-
|
|
2890
|
-
|
|
2891
|
-
}
|
|
2892
|
-
|
|
2893
|
-
|
|
2894
|
-
|
|
2895
|
-
|
|
2896
|
-
|
|
2897
|
-
|
|
2898
|
-
|
|
2899
|
-
* 1. plain `JSON.parse` of the opener → last-closer slice,
|
|
2900
|
-
* 2. auto-closing unclosed structures at the tail
|
|
2901
|
-
* (`autoCloseTruncatedJson`),
|
|
2902
|
-
* 3. trimming the tail back to the previous complete member boundary (the
|
|
2903
|
-
* last comma outside a string) and auto-closing again, repeatedly.
|
|
2904
|
-
*
|
|
2905
|
-
* Recovers e.g. `{"correct": false, "` → `{ correct: false }` — the exact
|
|
2906
|
-
* cap-hit shape that has zeroed real eval rows. Returns the parsed value
|
|
2907
|
-
* (always an object or array, given the slice starts at an opener), or
|
|
2908
|
-
* `null` when nothing parseable can be recovered. Never throws.
|
|
2909
|
-
*/
|
|
2910
|
-
function recoverTruncatedJson(text) {
|
|
2911
|
-
const starts = [text.indexOf("{"), text.indexOf("[")].filter((i) => i >= 0).sort((a, b) => a - b);
|
|
2912
|
-
for (const start of starts) {
|
|
2913
|
-
const recovered = recoverFrom(text.slice(start));
|
|
2914
|
-
if (recovered !== UNPARSEABLE) return recovered;
|
|
2915
|
-
}
|
|
2916
|
-
return null;
|
|
2917
|
-
}
|
|
2918
|
-
function recoverFrom(slice) {
|
|
2919
|
-
let candidate = slice;
|
|
2920
|
-
const lastClose = Math.max(candidate.lastIndexOf("}"), candidate.lastIndexOf("]"));
|
|
2921
|
-
if (lastClose > 0) {
|
|
2922
|
-
const balanced = tryParse(candidate.slice(0, lastClose + 1));
|
|
2923
|
-
if (balanced !== UNPARSEABLE) return balanced;
|
|
2924
|
-
}
|
|
2925
|
-
for (let i = 0; i < 64; i++) {
|
|
2926
|
-
const closed = autoCloseTruncatedJson(candidate);
|
|
2927
|
-
if (closed !== null) {
|
|
2928
|
-
const parsed = tryParse(closed);
|
|
2929
|
-
if (parsed !== UNPARSEABLE) return parsed;
|
|
2922
|
+
if (event.kind === "candidate-slot-closed") {
|
|
2923
|
+
const plannedSlot = plannedSlots.get(event.slotId);
|
|
2924
|
+
if (!plannedSlot) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} references unknown slot ${event.slotId}`);
|
|
2925
|
+
if (event.generationOperationId !== plannedSlot.generationOperationId) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} generation operation ${event.generationOperationId} does not match slot ${event.slotId} plan ${plannedSlot.generationOperationId}`);
|
|
2926
|
+
if (candidateBySlot.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was already bound to a candidate`);
|
|
2927
|
+
if (closedSlots.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was closed twice`);
|
|
2928
|
+
const operation = operationsById.get(event.generationOperationId);
|
|
2929
|
+
if (!operation) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} precedes operation ${event.generationOperationId}`);
|
|
2930
|
+
if (operation.outcome.status === "completed") throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} cannot close from completed operation ${event.generationOperationId}`);
|
|
2931
|
+
closedSlots.set(event.slotId, event);
|
|
2932
|
+
closedSlotEvents.push(event);
|
|
2933
|
+
return;
|
|
2930
2934
|
}
|
|
2931
|
-
|
|
2932
|
-
|
|
2933
|
-
|
|
2934
|
-
|
|
2935
|
-
|
|
2936
|
-
}
|
|
2937
|
-
|
|
2938
|
-
//#region src/reflective-mutation.ts
|
|
2939
|
-
/**
|
|
2940
|
-
* Reflective mutation — primitives for trace-conditioned prompt rewriting.
|
|
2941
|
-
*
|
|
2942
|
-
* Used by `prompt-evolution.ts` (and any consumer running iterative
|
|
2943
|
-
* improvement). Given a parent prompt + concrete trace evidence (top trials,
|
|
2944
|
-
* bottom trials, missed expectations), produce an LLM-ready prompt that
|
|
2945
|
-
* proposes targeted mutations — not blind rephrasings.
|
|
2946
|
-
*
|
|
2947
|
-
* Why this lives outside `prompt-evolution.ts`: any consumer that wants to
|
|
2948
|
-
* run reflective rewriting WITHOUT the population/Pareto machinery can
|
|
2949
|
-
* import these primitives directly.
|
|
2950
|
-
*
|
|
2951
|
-
* Quality bar (vs. naive "mutate this prompt"):
|
|
2952
|
-
* - Show parent ↔ children diff, not just one variant
|
|
2953
|
-
* - Quote specific missed goldens with their match phrases
|
|
2954
|
-
* - Surface the model's actual emitted output side-by-side with what was expected
|
|
2955
|
-
* - Quote concrete mutation primitives so the model has a vocabulary
|
|
2956
|
-
*/
|
|
2957
|
-
/** Bound on rendered/carried `emitted` evidence. ONE constant shared by the
|
|
2958
|
-
* producer (campaignBreakdown's per-scenario excerpt) and this renderer — if
|
|
2959
|
-
* the two drifted, the tighter side would silently re-clip carried evidence. */
|
|
2960
|
-
const EMITTED_EVIDENCE_MAX_CHARS = 2e3;
|
|
2961
|
-
const DEFAULT_MUTATION_PRIMITIVES = [
|
|
2962
|
-
"Strengthen an imperative (\"should\" → \"must\")",
|
|
2963
|
-
"Add a concrete example pulled from a missed-golden phrase",
|
|
2964
|
-
"Remove a redundant rule that did not improve recall",
|
|
2965
|
-
"Add a counterfactual (\"if X is missing, the score is capped at Y\")",
|
|
2966
|
-
"Reorder sections so the highest-impact rule is first",
|
|
2967
|
-
"Replace abstract language with a domain-specific noun the trial misses"
|
|
2968
|
-
];
|
|
2969
|
-
/**
|
|
2970
|
-
* Build the LLM-ready reflection prompt. Output is plain text — pass it as
|
|
2971
|
-
* the user message. The system message should be small and stable (e.g.
|
|
2972
|
-
* "Output ONLY a JSON object matching the schema below.").
|
|
2973
|
-
*/
|
|
2974
|
-
function buildReflectionPrompt(ctx) {
|
|
2975
|
-
const primitives = ctx.mutationPrimitives ?? DEFAULT_MUTATION_PRIMITIVES;
|
|
2976
|
-
const sections = [];
|
|
2977
|
-
sections.push(`# Mutation target: ${ctx.target}`);
|
|
2978
|
-
sections.push("");
|
|
2979
|
-
sections.push(`You are tuning the prompt component named \`${ctx.target}\`. The current variant is shown below; you have ${ctx.topTrials.length} top trials and ${ctx.bottomTrials.length} bottom trials as evidence. Propose ${ctx.childCount} mutation${ctx.childCount === 1 ? "" : "s"} that fix specific weaknesses visible in the bottom trials. Avoid blank rephrasings.`);
|
|
2980
|
-
sections.push("");
|
|
2981
|
-
sections.push("## Current variant");
|
|
2982
|
-
sections.push("```json");
|
|
2983
|
-
sections.push(JSON.stringify(ctx.parentPayload, null, 2));
|
|
2984
|
-
sections.push("```");
|
|
2985
|
-
sections.push("");
|
|
2986
|
-
if (ctx.bottomTrials.length > 0) {
|
|
2987
|
-
sections.push("## Failures (bottom trials) — what went wrong");
|
|
2988
|
-
sections.push("");
|
|
2989
|
-
for (const trial of ctx.bottomTrials) {
|
|
2990
|
-
sections.push(`### Trial \`${trial.id}\` — score ${trial.score.toFixed(2)}${trial.inputName ? ` (${trial.inputName})` : ""}`);
|
|
2991
|
-
if (trial.failureNote) {
|
|
2992
|
-
sections.push("");
|
|
2993
|
-
sections.push(`**Why it scored low:** ${truncate(trial.failureNote, 1500)}`);
|
|
2994
|
-
}
|
|
2995
|
-
const missed = (trial.expectations ?? []).filter((e) => !e.matched);
|
|
2996
|
-
if (missed.length > 0) {
|
|
2997
|
-
sections.push("");
|
|
2998
|
-
sections.push("**Missed expectations:**");
|
|
2999
|
-
for (const m of missed) sections.push(`- \`${m.id}\`: should match phrase \`${quote(m.phrase)}\``);
|
|
2935
|
+
if (event.kind === "candidate-decided") {
|
|
2936
|
+
const candidate = candidates.get(event.candidateId);
|
|
2937
|
+
if (!candidate) throw new SearchLedgerIntegrityError(`decision ${event.eventId} references unknown candidate ${event.candidateId}`);
|
|
2938
|
+
if (candidate.decision) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} was decided twice`);
|
|
2939
|
+
if (event.decision.status === "selected") {
|
|
2940
|
+
if (!candidate.attempts.some((attempt) => attempt.outcome.status !== "errored")) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} cannot be selected without a measured task outcome`);
|
|
2941
|
+
if (decisions.some((decision) => decision.decision.status === "selected")) throw new SearchLedgerIntegrityError("more than one candidate was selected");
|
|
3000
2942
|
}
|
|
3001
|
-
|
|
3002
|
-
|
|
3003
|
-
|
|
3004
|
-
|
|
3005
|
-
|
|
3006
|
-
|
|
2943
|
+
candidate.decision = event;
|
|
2944
|
+
decisions.push(event);
|
|
2945
|
+
return;
|
|
2946
|
+
}
|
|
2947
|
+
const missingCandidateSlots = [...plannedSlots.values()].filter((slot) => !candidateBySlot.has(slot.slotId) && !closedSlots.has(slot.slotId)).map((slot) => slot.slotId);
|
|
2948
|
+
if (missingCandidateSlots.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing candidate slots: ${missingCandidateSlots.join(", ")}`);
|
|
2949
|
+
const missingTaskOutcomes = plannedTaskOutcomeKeys(planEvent, candidates);
|
|
2950
|
+
if (missingTaskOutcomes.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing task outcomes: ${missingTaskOutcomes.join(", ")}`);
|
|
2951
|
+
const missingOperations = [...plannedOperations.values()].filter((operation) => !operationsById.has(operation.operationId)).map((operation) => operation.operationId);
|
|
2952
|
+
if (missingOperations.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing search operations: ${missingOperations.join(", ")}`);
|
|
2953
|
+
for (const operation of plannedOperations.values()) {
|
|
2954
|
+
if (operation.kind !== "candidate-generation") continue;
|
|
2955
|
+
const generationOutcome = operationsById.get(operation.operationId).outcome.status;
|
|
2956
|
+
const slots = [...plannedSlots.values()].filter((slot) => slot.generationOperationId === operation.operationId);
|
|
2957
|
+
if (slots.length === 0) continue;
|
|
2958
|
+
const registeredCount = slots.filter((slot) => candidateBySlot.has(slot.slotId)).length;
|
|
2959
|
+
const closedCount = slots.filter((slot) => closedSlots.has(slot.slotId)).length;
|
|
2960
|
+
if (generationOutcome === "completed" && closedCount > 0) throw new SearchLedgerIntegrityError(`completed generation operation ${operation.operationId} contains ${closedCount} closed slot(s)`);
|
|
2961
|
+
if (generationOutcome === "failed" && registeredCount > 0) throw new SearchLedgerIntegrityError(`failed generation operation ${operation.operationId} contains ${registeredCount} registered candidate(s)`);
|
|
2962
|
+
if (generationOutcome === "partial" && (registeredCount === 0 || closedCount === 0)) throw new SearchLedgerIntegrityError(`partial generation operation ${operation.operationId} must contain both a registered candidate and a closed slot`);
|
|
2963
|
+
}
|
|
2964
|
+
const pending = [...candidates.values()].filter((candidate) => candidate.decision === null);
|
|
2965
|
+
if (pending.length > 0) throw new SearchLedgerIntegrityError(`search completed with ${pending.length} candidate decision(s) missing`);
|
|
2966
|
+
const selected = decisions.filter((decision) => decision.decision.status === "selected");
|
|
2967
|
+
if (event.result.status === "selected") {
|
|
2968
|
+
if (selected.length !== 1 || selected[0].candidateId !== event.result.candidateId) throw new SearchLedgerIntegrityError(`search completion winner ${event.result.candidateId} does not match candidate decisions`);
|
|
2969
|
+
} else if (selected.length !== 0) throw new SearchLedgerIntegrityError("all-rejected completion contains a selected candidate");
|
|
2970
|
+
completion = event;
|
|
2971
|
+
};
|
|
2972
|
+
const finish = (entries) => {
|
|
2973
|
+
const selectedDecisions = decisions.filter((decision) => decision.decision.status === "selected");
|
|
2974
|
+
const rejectedDecisions = decisions.filter((decision) => decision.decision.status === "rejected");
|
|
2975
|
+
const outcomeCounts = {
|
|
2976
|
+
passed: 0,
|
|
2977
|
+
failed: 0,
|
|
2978
|
+
errored: 0
|
|
2979
|
+
};
|
|
2980
|
+
const operationOutcomeCounts = {
|
|
2981
|
+
completed: 0,
|
|
2982
|
+
partial: 0,
|
|
2983
|
+
failed: 0
|
|
2984
|
+
};
|
|
2985
|
+
let inputTokens = 0;
|
|
2986
|
+
let outputTokens = 0;
|
|
2987
|
+
let cachedTokens = 0;
|
|
2988
|
+
let costUsd = 0;
|
|
2989
|
+
const unknownTokenEventIds = [];
|
|
2990
|
+
const unknownCostEventIds = [];
|
|
2991
|
+
for (const attempt of attempts) outcomeCounts[attempt.outcome.status] += 1;
|
|
2992
|
+
for (const operation of operationEvents) operationOutcomeCounts[operation.outcome.status] += 1;
|
|
2993
|
+
for (const costedEvent of [...attempts, ...operationEvents]) {
|
|
2994
|
+
if (costedEvent.accounting.tokens.status === "known") {
|
|
2995
|
+
inputTokens += costedEvent.accounting.tokens.inputTokens;
|
|
2996
|
+
outputTokens += costedEvent.accounting.tokens.outputTokens;
|
|
2997
|
+
cachedTokens += costedEvent.accounting.tokens.cachedTokens;
|
|
2998
|
+
} else unknownTokenEventIds.push(costedEvent.eventId);
|
|
2999
|
+
if (costedEvent.accounting.cost.status === "known") costUsd += costedEvent.accounting.cost.usd;
|
|
3000
|
+
else {
|
|
3001
|
+
costUsd += costedEvent.accounting.cost.knownLowerBoundUsd;
|
|
3002
|
+
unknownCostEventIds.push(costedEvent.eventId);
|
|
3007
3003
|
}
|
|
3008
|
-
sections.push("");
|
|
3009
3004
|
}
|
|
3010
|
-
|
|
3011
|
-
|
|
3012
|
-
|
|
3013
|
-
|
|
3014
|
-
|
|
3015
|
-
|
|
3016
|
-
|
|
3017
|
-
|
|
3018
|
-
|
|
3019
|
-
|
|
3020
|
-
|
|
3021
|
-
|
|
3022
|
-
|
|
3023
|
-
|
|
3024
|
-
|
|
3025
|
-
|
|
3026
|
-
|
|
3027
|
-
|
|
3028
|
-
|
|
3029
|
-
|
|
3030
|
-
|
|
3031
|
-
|
|
3032
|
-
|
|
3033
|
-
|
|
3034
|
-
|
|
3035
|
-
|
|
3036
|
-
|
|
3037
|
-
|
|
3038
|
-
|
|
3005
|
+
const accounting = unknownTokenEventIds.length === 0 && unknownCostEventIds.length === 0 ? {
|
|
3006
|
+
status: "known",
|
|
3007
|
+
inputTokens,
|
|
3008
|
+
outputTokens,
|
|
3009
|
+
cachedTokens,
|
|
3010
|
+
costUsd
|
|
3011
|
+
} : {
|
|
3012
|
+
status: "partial",
|
|
3013
|
+
knownInputTokens: inputTokens,
|
|
3014
|
+
knownOutputTokens: outputTokens,
|
|
3015
|
+
knownCachedTokens: cachedTokens,
|
|
3016
|
+
knownCostUsd: costUsd,
|
|
3017
|
+
unknownTokenEventIds,
|
|
3018
|
+
unknownCostEventIds
|
|
3019
|
+
};
|
|
3020
|
+
const selectedCandidateId = completion?.result.status === "selected" ? completion.result.candidateId : null;
|
|
3021
|
+
const status = completion?.result.status === "selected" ? "selected" : completion?.result.status === "all-rejected" ? "all-rejected" : "in-progress";
|
|
3022
|
+
const missingCandidateSlots = [...plannedSlots.values()].filter((slot) => !candidateBySlot.has(slot.slotId) && !closedSlots.has(slot.slotId)).map((slot) => slot.slotId);
|
|
3023
|
+
const missingTaskOutcomes = planEvent ? plannedTaskOutcomeKeys(planEvent, candidates) : [];
|
|
3024
|
+
const missingOperations = [...plannedOperations.values()].filter((operation) => !operationsById.has(operation.operationId)).map((operation) => operation.operationId);
|
|
3025
|
+
return {
|
|
3026
|
+
entries: [...entries],
|
|
3027
|
+
plan: planEvent,
|
|
3028
|
+
planExtensions,
|
|
3029
|
+
candidates: candidateEvents,
|
|
3030
|
+
closedCandidateSlots: closedSlotEvents,
|
|
3031
|
+
attempts,
|
|
3032
|
+
operations: operationEvents,
|
|
3033
|
+
decisions,
|
|
3034
|
+
completion,
|
|
3035
|
+
audit: {
|
|
3036
|
+
campaignId,
|
|
3037
|
+
eventCount: entries.length,
|
|
3038
|
+
candidateCount: candidates.size,
|
|
3039
|
+
closedCandidateSlotCount: closedSlots.size,
|
|
3040
|
+
attemptCount: attempts.length,
|
|
3041
|
+
operationCount: operationEvents.length,
|
|
3042
|
+
outcomes: outcomeCounts,
|
|
3043
|
+
operationOutcomes: operationOutcomeCounts,
|
|
3044
|
+
decisions: {
|
|
3045
|
+
selected: selectedDecisions.length,
|
|
3046
|
+
rejected: rejectedDecisions.length,
|
|
3047
|
+
pending: candidates.size - decisions.length
|
|
3048
|
+
},
|
|
3049
|
+
expected: {
|
|
3050
|
+
candidateSlots: plannedSlots.size,
|
|
3051
|
+
taskOutcomes: candidates.size * (planEvent?.plan.tasks.length ?? 0),
|
|
3052
|
+
operations: plannedOperations.size,
|
|
3053
|
+
missingCandidateSlots,
|
|
3054
|
+
missingTaskOutcomes,
|
|
3055
|
+
missingOperations
|
|
3056
|
+
},
|
|
3057
|
+
status,
|
|
3058
|
+
selectedCandidateId,
|
|
3059
|
+
accounting,
|
|
3060
|
+
headHash: entries.at(-1)?.entryHash ?? null
|
|
3061
|
+
}
|
|
3062
|
+
};
|
|
3063
|
+
};
|
|
3064
|
+
return {
|
|
3065
|
+
apply,
|
|
3066
|
+
finish
|
|
3067
|
+
};
|
|
3039
3068
|
}
|
|
3040
|
-
/**
|
|
3041
|
-
*
|
|
3042
|
-
|
|
3043
|
-
|
|
3044
|
-
function parseReflectionResponse(raw, maxProposals) {
|
|
3045
|
-
let text = raw.trim();
|
|
3046
|
-
if (text.startsWith("```")) text = text.replace(/^```(?:json)?\n?/, "").replace(/\n?```$/, "");
|
|
3047
|
-
let parsed = null;
|
|
3048
|
-
const objectStart = text.indexOf("{");
|
|
3049
|
-
const objectEnd = text.lastIndexOf("}");
|
|
3050
|
-
const arrayStart = text.indexOf("[");
|
|
3051
|
-
const arrayEnd = text.lastIndexOf("]");
|
|
3052
|
-
const tryObjectFirst = objectStart >= 0 && (arrayStart < 0 || objectStart < arrayStart);
|
|
3053
|
-
const candidates = [];
|
|
3054
|
-
if (tryObjectFirst) {
|
|
3055
|
-
if (objectStart >= 0 && objectEnd > objectStart) candidates.push(text.slice(objectStart, objectEnd + 1));
|
|
3056
|
-
if (arrayStart >= 0 && arrayEnd > arrayStart) candidates.push(text.slice(arrayStart, arrayEnd + 1));
|
|
3057
|
-
} else {
|
|
3058
|
-
if (arrayStart >= 0 && arrayEnd > arrayStart) candidates.push(text.slice(arrayStart, arrayEnd + 1));
|
|
3059
|
-
if (objectStart >= 0 && objectEnd > objectStart) candidates.push(text.slice(objectStart, objectEnd + 1));
|
|
3060
|
-
}
|
|
3061
|
-
for (const slice of candidates) try {
|
|
3062
|
-
parsed = JSON.parse(slice);
|
|
3063
|
-
break;
|
|
3064
|
-
} catch {}
|
|
3065
|
-
if (parsed == null) for (const slice of candidates) {
|
|
3066
|
-
const closed = autoCloseTruncatedJson(slice);
|
|
3067
|
-
if (closed != null && closed !== slice) try {
|
|
3068
|
-
parsed = JSON.parse(closed);
|
|
3069
|
-
break;
|
|
3070
|
-
} catch {}
|
|
3071
|
-
}
|
|
3072
|
-
if (parsed == null) return [];
|
|
3073
|
-
let proposalsRaw;
|
|
3074
|
-
if (Array.isArray(parsed)) proposalsRaw = parsed;
|
|
3075
|
-
else if (parsed && typeof parsed === "object") proposalsRaw = parsed.proposals;
|
|
3076
|
-
if (!Array.isArray(proposalsRaw)) return [];
|
|
3077
|
-
const out = [];
|
|
3078
|
-
for (const p of proposalsRaw) {
|
|
3079
|
-
if (!p || typeof p !== "object") continue;
|
|
3080
|
-
const obj = p;
|
|
3081
|
-
if (!("payload" in obj)) continue;
|
|
3082
|
-
out.push({
|
|
3083
|
-
label: typeof obj.label === "string" ? obj.label : "mutation",
|
|
3084
|
-
rationale: typeof obj.rationale === "string" ? obj.rationale : "",
|
|
3085
|
-
payload: obj.payload
|
|
3086
|
-
});
|
|
3087
|
-
if (maxProposals !== void 0 && out.length >= maxProposals) break;
|
|
3088
|
-
}
|
|
3089
|
-
return out;
|
|
3069
|
+
/** Every candidate slot must name a planned candidate-generation operation,
|
|
3070
|
+
* whether it arrives with the plan or with a later extension. */
|
|
3071
|
+
function assertSlotGenerationOperation(slot, plannedOperations) {
|
|
3072
|
+
if (plannedOperations.get(slot.generationOperationId)?.kind !== "candidate-generation") throw new SearchLedgerIntegrityError(`candidate slot ${slot.slotId} references unplanned candidate-generation operation ${slot.generationOperationId}`);
|
|
3090
3073
|
}
|
|
3091
|
-
|
|
3092
|
-
|
|
3093
|
-
|
|
3094
|
-
|
|
3095
|
-
|
|
3096
|
-
|
|
3097
|
-
* the optimizers cannot drift on how a surface's score is computed.
|
|
3098
|
-
*/
|
|
3099
|
-
/** Mean composite across cells with complete task-quality evidence.
|
|
3100
|
-
* Partial judge results remain on their cells but never enter this value.
|
|
3101
|
-
* A campaign with no complete score has no numeric mean and fails loudly. */
|
|
3102
|
-
function campaignMeanComposite(campaign) {
|
|
3103
|
-
const mean = campaignMeanCompositeOrNull(campaign);
|
|
3104
|
-
if (mean === null) throw new Error("campaignMeanComposite: campaign has no complete cell-quality scores");
|
|
3105
|
-
return mean;
|
|
3106
|
-
}
|
|
3107
|
-
/** Nullable campaign mean for wire and report fields that represent missing quality. */
|
|
3108
|
-
function campaignMeanCompositeOrNull(campaign) {
|
|
3109
|
-
const scores = campaign.cells.flatMap((cell) => {
|
|
3110
|
-
const score = projectCampaignCellQuality(cell).score;
|
|
3111
|
-
return score === void 0 ? [] : [score];
|
|
3112
|
-
});
|
|
3113
|
-
return scores.length === 0 ? null : scores.reduce((sum, score) => sum + score, 0) / scores.length;
|
|
3114
|
-
}
|
|
3115
|
-
/** Reject rank keys that cannot produce deterministic lexicographic ordering. */
|
|
3116
|
-
function assertFiniteRankKey(key, label, expectedLength) {
|
|
3117
|
-
if (!Array.isArray(key) || key.length === 0) throw new Error(`${label} must return a non-empty array`);
|
|
3118
|
-
if (expectedLength !== void 0 && key.length !== expectedLength) throw new Error(`${label} returned ${key.length} elements; expected ${expectedLength}`);
|
|
3119
|
-
for (let index = 0; index < key.length; index++) if (!Number.isFinite(key[index])) throw new Error(`${label}[${index}] must be finite`);
|
|
3120
|
-
}
|
|
3121
|
-
/** Compare fixed-length lexicographic rank keys where each element is higher-is-better.
|
|
3122
|
-
* Returns a positive number when `a` ranks above `b`, negative when below, and
|
|
3123
|
-
* zero when equal. */
|
|
3124
|
-
function compareRankKeys(a, b) {
|
|
3125
|
-
assertFiniteRankKey(a, "rank key a");
|
|
3126
|
-
assertFiniteRankKey(b, "rank key b", a.length);
|
|
3127
|
-
for (let i = 0; i < a.length; i++) {
|
|
3128
|
-
const av = a[i];
|
|
3129
|
-
const bv = b[i];
|
|
3130
|
-
if (av !== bv) return av - bv;
|
|
3074
|
+
function plannedTaskOutcomeKeys(planEvent, candidates) {
|
|
3075
|
+
const missing = [];
|
|
3076
|
+
const registeredCandidates = [...candidates.values()].sort((a, b) => compareStrings(a.registered.slotId, b.registered.slotId));
|
|
3077
|
+
for (const candidate of registeredCandidates) {
|
|
3078
|
+
const slotId = candidate.registered.slotId;
|
|
3079
|
+
for (const task of planEvent.plan.tasks) if (!candidate.attempts.some((attempt) => attempt.task.taskId === task.taskId && attempt.outcome.status !== "errored")) missing.push(`${slotId}/${task.taskId}`);
|
|
3131
3080
|
}
|
|
3132
|
-
return
|
|
3133
|
-
}
|
|
3134
|
-
|
|
3135
|
-
|
|
3136
|
-
|
|
3137
|
-
|
|
3138
|
-
|
|
3139
|
-
|
|
3140
|
-
|
|
3141
|
-
|
|
3142
|
-
|
|
3143
|
-
const quality = projectCampaignCellQuality(cell);
|
|
3144
|
-
if (quality.score === void 0) continue;
|
|
3145
|
-
const judgeScores = Object.values(quality.successfulJudgeScores);
|
|
3146
|
-
const cellComposite = quality.score;
|
|
3147
|
-
const arr = byScenario.get(cell.scenarioId) ?? [];
|
|
3148
|
-
arr.push(cellComposite);
|
|
3149
|
-
byScenario.set(cell.scenarioId, arr);
|
|
3150
|
-
if (typeof cell.artifact === "string" && cell.artifact.trim().length > 0) {
|
|
3151
|
-
const prev = emittedByScenario.get(cell.scenarioId);
|
|
3152
|
-
if (!prev || cellComposite < prev.composite) emittedByScenario.set(cell.scenarioId, {
|
|
3153
|
-
composite: cellComposite,
|
|
3154
|
-
text: cell.artifact.slice(0, EMITTED_EVIDENCE_MAX_CHARS)
|
|
3155
|
-
});
|
|
3156
|
-
}
|
|
3157
|
-
for (const s of judgeScores) if (s.notes?.trim()) {
|
|
3158
|
-
const set = notesByScenario.get(cell.scenarioId) ?? /* @__PURE__ */ new Set();
|
|
3159
|
-
set.add(s.notes.trim());
|
|
3160
|
-
notesByScenario.set(cell.scenarioId, set);
|
|
3081
|
+
return missing;
|
|
3082
|
+
}
|
|
3083
|
+
function normalizeEvent(event) {
|
|
3084
|
+
const artifacts = sortArtifacts(event.artifacts);
|
|
3085
|
+
if (event.kind === "search-planned") return {
|
|
3086
|
+
...event,
|
|
3087
|
+
artifacts,
|
|
3088
|
+
plan: {
|
|
3089
|
+
candidateSlots: [...event.plan.candidateSlots].sort((a, b) => compareStrings(a.slotId, b.slotId)),
|
|
3090
|
+
tasks: [...event.plan.tasks].sort((a, b) => compareStrings(a.taskId, b.taskId)),
|
|
3091
|
+
operations: [...event.plan.operations].sort((a, b) => compareStrings(a.operationId, b.operationId))
|
|
3161
3092
|
}
|
|
3162
|
-
|
|
3163
|
-
|
|
3164
|
-
|
|
3165
|
-
|
|
3093
|
+
};
|
|
3094
|
+
if (event.kind === "search-plan-extended") return {
|
|
3095
|
+
...event,
|
|
3096
|
+
artifacts,
|
|
3097
|
+
extension: {
|
|
3098
|
+
candidateSlots: [...event.extension.candidateSlots].sort((a, b) => compareStrings(a.slotId, b.slotId)),
|
|
3099
|
+
operations: [...event.extension.operations].sort((a, b) => compareStrings(a.operationId, b.operationId))
|
|
3166
3100
|
}
|
|
3167
|
-
}
|
|
3168
|
-
|
|
3169
|
-
|
|
3170
|
-
|
|
3171
|
-
|
|
3172
|
-
|
|
3101
|
+
};
|
|
3102
|
+
if (event.kind === "candidate-registered") return {
|
|
3103
|
+
...event,
|
|
3104
|
+
artifacts,
|
|
3105
|
+
lineage: {
|
|
3106
|
+
...event.lineage,
|
|
3107
|
+
parentCandidateIds: sortedStrings(event.lineage.parentCandidateIds)
|
|
3108
|
+
},
|
|
3109
|
+
surfaces: [...event.surfaces].map((surface) => ({
|
|
3110
|
+
...surface,
|
|
3111
|
+
artifact: { ...surface.artifact }
|
|
3112
|
+
})).sort((a, b) => compareStrings(a.surfaceId, b.surfaceId))
|
|
3113
|
+
};
|
|
3114
|
+
if (event.kind === "task-attempted") return {
|
|
3115
|
+
...event,
|
|
3116
|
+
artifacts,
|
|
3117
|
+
surfaceEvidence: [...event.surfaceEvidence].map((evidence) => ({
|
|
3118
|
+
...evidence,
|
|
3119
|
+
evidence: sortArtifacts(evidence.evidence)
|
|
3120
|
+
})).sort((a, b) => compareStrings(a.surfaceId, b.surfaceId))
|
|
3121
|
+
};
|
|
3173
3122
|
return {
|
|
3174
|
-
|
|
3175
|
-
|
|
3176
|
-
const notesSet = notesByScenario.get(scenarioId);
|
|
3177
|
-
const notes = notesSet && notesSet.size > 0 ? [...notesSet].join(" | ") : void 0;
|
|
3178
|
-
const emitted = emittedByScenario.get(scenarioId)?.text;
|
|
3179
|
-
return {
|
|
3180
|
-
scenarioId,
|
|
3181
|
-
composite: comps.reduce((a, b) => a + b, 0) / comps.length,
|
|
3182
|
-
...notes ? { notes } : {},
|
|
3183
|
-
...emitted ? { emitted } : {}
|
|
3184
|
-
};
|
|
3185
|
-
})
|
|
3123
|
+
...event,
|
|
3124
|
+
artifacts
|
|
3186
3125
|
};
|
|
3187
3126
|
}
|
|
3127
|
+
function sortArtifacts(artifacts) {
|
|
3128
|
+
return [...artifacts].map((artifact) => ({ ...artifact })).sort((a, b) => compareStrings(artifactKey(a), artifactKey(b)));
|
|
3129
|
+
}
|
|
3130
|
+
function artifactKey(artifact) {
|
|
3131
|
+
return canonicalString(artifact);
|
|
3132
|
+
}
|
|
3133
|
+
function compareStrings(a, b) {
|
|
3134
|
+
return a < b ? -1 : a > b ? 1 : 0;
|
|
3135
|
+
}
|
|
3136
|
+
function sortedStrings(values) {
|
|
3137
|
+
return [...values].sort();
|
|
3138
|
+
}
|
|
3139
|
+
function assertUnique(values, label, eventId) {
|
|
3140
|
+
if (new Set(values).size !== values.length) throw new SearchLedgerIntegrityError(`event ${eventId} contains duplicate ${label} values`);
|
|
3141
|
+
}
|
|
3142
|
+
function formatZodError(error) {
|
|
3143
|
+
return error.issues.map((issue) => `${issue.path.length > 0 ? issue.path.join(".") : "<root>"}: ${issue.message}`).join("; ");
|
|
3144
|
+
}
|
|
3188
3145
|
//#endregion
|
|
3189
3146
|
//#region src/campaign/search-history-receipt.ts
|
|
3190
3147
|
const SEARCH_HISTORY_RECEIPT_SCHEMA_VERSION = "1.0.0";
|
|
3191
3148
|
const SEARCH_HISTORY_RECEIPT_DIGEST_ALGORITHM = "rfc8785-sha256";
|
|
3149
|
+
function assertSearchHistoryAdmissionOptions(options) {
|
|
3150
|
+
if (options.searchHistoryPolicy !== void 0 && !["allow-missing", "require-complete"].includes(options.searchHistoryPolicy)) throw new Error(`unknown searchHistoryPolicy '${String(options.searchHistoryPolicy)}'`);
|
|
3151
|
+
if (options.searchHistoryVerification !== void 0 && !["receipt", "ledger"].includes(options.searchHistoryVerification)) throw new Error(`unknown searchHistoryVerification '${String(options.searchHistoryVerification)}'`);
|
|
3152
|
+
}
|
|
3153
|
+
/** Verify resolved bytes with the canonical codec, never a caller-supplied replay projection. */
|
|
3154
|
+
function verifySearchHistoryArtifact(receipt, storage) {
|
|
3155
|
+
verifySearchHistoryReceipt(receipt);
|
|
3156
|
+
const path = receipt.ledger.uri.startsWith("file:") ? fileURLToPath(receipt.ledger.uri) : receipt.ledger.uri;
|
|
3157
|
+
const text = storage.read(path);
|
|
3158
|
+
if (text === void 0) throw new Error(`search history ledger is missing or unreadable at '${path}'`);
|
|
3159
|
+
if (new TextEncoder().encode(text).byteLength !== receipt.ledger.byteLength) throw new Error("search history ledger byte length mismatch");
|
|
3160
|
+
if (hashCanonical(text) !== receipt.ledger.sha256) throw new Error("search history ledger digest mismatch");
|
|
3161
|
+
assertSearchHistoryMatchesReplay(receipt, replaySearchLedgerText(text, receipt.summary.campaignId, path));
|
|
3162
|
+
}
|
|
3192
3163
|
var SearchHistoryRequiredError = class extends Error {
|
|
3193
3164
|
producerId;
|
|
3194
3165
|
reasons;
|
|
@@ -4696,666 +4667,23 @@ async function runImprovementLoop(opts) {
|
|
|
4696
4667
|
};
|
|
4697
4668
|
}
|
|
4698
4669
|
//#endregion
|
|
4699
|
-
//#region src/campaign/
|
|
4670
|
+
//#region src/campaign/gepa-candidate-population.ts
|
|
4700
4671
|
/**
|
|
4701
|
-
*
|
|
4702
|
-
* loop did and WHY, plus the OTel spans that let an OTLP collector pivot from
|
|
4703
|
-
* an eval-run to the underlying candidate→cell→gate→promote chain.
|
|
4704
|
-
*
|
|
4705
|
-
* Two artifacts, one source of truth:
|
|
4706
|
-
*
|
|
4707
|
-
* 1. `LoopProvenanceRecord` — a structured JSON record capturing every
|
|
4708
|
-
* candidate (surfaceHash + label + rationale + structured cause), its measured composite,
|
|
4709
|
-
* the gate decision + reasons + delta, the held-out lift, the explicit
|
|
4710
|
-
* baseline→candidate diff, and BACKEND PROVENANCE (the
|
|
4711
|
-
* `assertRealBackend` verdict + worker call count + model). This is the
|
|
4712
|
-
* ingestable audit artifact: the +lift recomputes from it, the "because
|
|
4713
|
-
* Z" rationale survives in it, and a stub backend is detectable from it.
|
|
4714
|
-
*
|
|
4715
|
-
* 2. `loopProvenanceSpans()` — the same chain emitted as OTLP-ingestable
|
|
4716
|
-
* `TraceSpanEvent`s, pivoted on the substrate's standard
|
|
4717
|
-
* `tangle.runId` / `tangle.scenarioId` / `tangle.cellId` /
|
|
4718
|
-
* `tangle.generation` attributes (the same pivots `/adapters/otel`
|
|
4719
|
-
* reads). The hosted `/v1/ingest/traces` endpoint receives the FULL loop,
|
|
4720
|
-
* not just the `cost.*` spans `runCampaign` already emits per cell.
|
|
4672
|
+
* Read GEPA's exact candidate graph from the artifact addressed by method provenance.
|
|
4721
4673
|
*
|
|
4722
|
-
* The
|
|
4723
|
-
*
|
|
4674
|
+
* The reader checks the supplied digest, declared byte count, run identity,
|
|
4675
|
+
* candidate surfaces, parent graph, selection scores, and configured bounds.
|
|
4676
|
+
* This proves that the bytes match the supplied summary. The caller remains
|
|
4677
|
+
* responsible for obtaining that summary from trusted method provenance.
|
|
4724
4678
|
*/
|
|
4725
|
-
|
|
4726
|
-
|
|
4727
|
-
const
|
|
4728
|
-
|
|
4729
|
-
|
|
4730
|
-
|
|
4731
|
-
|
|
4732
|
-
|
|
4733
|
-
winnerSurface: result.winnerSurface,
|
|
4734
|
-
...result.winnerLabel ? { winnerLabel: result.winnerLabel } : {},
|
|
4735
|
-
...result.winnerRationale ? { winnerRationale: result.winnerRationale } : {},
|
|
4736
|
-
baselineSearchCampaign: result.baselineCampaign,
|
|
4737
|
-
generations: result.generations.map(({ record, surfaces }) => ({
|
|
4738
|
-
generationIndex: record.generationIndex,
|
|
4739
|
-
candidates: record.candidates,
|
|
4740
|
-
promoted: record.promoted,
|
|
4741
|
-
surfaces: surfaces.map(({ surfaceHash, surface, campaign }) => ({
|
|
4742
|
-
surfaceHash,
|
|
4743
|
-
surface,
|
|
4744
|
-
campaign
|
|
4745
|
-
}))
|
|
4746
|
-
})),
|
|
4747
|
-
gate: result.gateResult,
|
|
4748
|
-
...result.holdout === "deferred" ? { holdout: "deferred" } : {},
|
|
4749
|
-
baselineOnHoldout: result.baselineOnHoldout,
|
|
4750
|
-
winnerOnHoldout: result.winnerOnHoldout,
|
|
4751
|
-
...result.neutralizedSurface && result.neutralizedOnHoldout ? {
|
|
4752
|
-
neutralizedSurface: result.neutralizedSurface,
|
|
4753
|
-
neutralizedOnHoldout: result.neutralizedOnHoldout
|
|
4754
|
-
} : {},
|
|
4755
|
-
costReceipts: input.costReceipts,
|
|
4756
|
-
totalCostUsd: input.totalCostUsd,
|
|
4757
|
-
totalDurationMs: input.totalDurationMs
|
|
4758
|
-
};
|
|
4759
|
-
}
|
|
4760
|
-
function meanHoldoutComposite(campaign) {
|
|
4761
|
-
return campaignMeanComposite(campaign);
|
|
4762
|
-
}
|
|
4763
|
-
/** Build the durable provenance record from a completed loop result. */
|
|
4764
|
-
function buildLoopProvenanceRecord(args) {
|
|
4765
|
-
if (!args.runId.trim() || !args.runDir.trim()) throw new Error("buildLoopProvenanceRecord: runId and runDir must be non-empty");
|
|
4766
|
-
const timestampMs = Date.parse(args.timestamp);
|
|
4767
|
-
if (!Number.isFinite(timestampMs) || new Date(timestampMs).toISOString() !== args.timestamp) throw new Error("buildLoopProvenanceRecord: timestamp must be a canonical ISO instant");
|
|
4768
|
-
assertGateContributions(args.gate.contributingGates, "buildLoopProvenanceRecord");
|
|
4769
|
-
const agentReceipts = args.costReceipts.filter((receipt) => receipt.channel === "agent");
|
|
4770
|
-
const integrity = summarizeAgentReceiptIntegrity(agentReceipts);
|
|
4771
|
-
const models = [...new Set(agentReceipts.map((receipt) => receipt.model))].sort();
|
|
4772
|
-
const baselineSearchComposite = campaignMeanComposite(args.baselineSearchCampaign);
|
|
4773
|
-
if (!Number.isFinite(baselineSearchComposite)) throw new Error("buildLoopProvenanceRecord: baselineSearchComposite must be finite");
|
|
4774
|
-
const candidates = [];
|
|
4775
|
-
let incumbentSurfaceHash = surfaceHash(args.baselineSurface);
|
|
4776
|
-
let incumbentComposite = baselineSearchComposite;
|
|
4777
|
-
let previousGeneration = -1;
|
|
4778
|
-
for (const gen of args.generations) {
|
|
4779
|
-
if (!Number.isSafeInteger(gen.generationIndex) || gen.generationIndex !== previousGeneration + 1) throw new Error("buildLoopProvenanceRecord: generation indices must be contiguous integers starting at zero");
|
|
4780
|
-
previousGeneration = gen.generationIndex;
|
|
4781
|
-
if (gen.candidates.length === 0) throw new Error("buildLoopProvenanceRecord: a recorded generation must contain a candidate");
|
|
4782
|
-
if (new Set(gen.promoted).size !== gen.promoted.length || gen.promoted.length > 1) throw new Error("buildLoopProvenanceRecord: each generation may promote at most one candidate");
|
|
4783
|
-
const promotedSet = new Set(gen.promoted);
|
|
4784
|
-
const surfaceByHash = new Map(gen.surfaces.map((measured) => [measured.surfaceHash, measured]));
|
|
4785
|
-
const candidateByHash = new Map(gen.candidates.map((candidate) => [candidate.surfaceHash, candidate]));
|
|
4786
|
-
if (candidateByHash.size !== gen.candidates.length) throw new Error("buildLoopProvenanceRecord: duplicate candidate surface hash");
|
|
4787
|
-
if (surfaceByHash.size !== gen.surfaces.length) throw new Error("buildLoopProvenanceRecord: duplicate candidate surface entry");
|
|
4788
|
-
if (surfaceByHash.size !== candidateByHash.size) throw new Error("buildLoopProvenanceRecord: every measured candidate requires exactly one surface");
|
|
4789
|
-
for (const promotedHash of promotedSet) if (!candidateByHash.has(promotedHash)) throw new Error("buildLoopProvenanceRecord: promoted hash has no measured candidate");
|
|
4790
|
-
for (const c of gen.candidates) {
|
|
4791
|
-
validateCandidateMeasurement(c, incumbentSurfaceHash, incumbentComposite, promotedSet.has(c.surfaceHash));
|
|
4792
|
-
const measured = surfaceByHash.get(c.surfaceHash);
|
|
4793
|
-
if (measured === void 0) throw new Error("buildLoopProvenanceRecord: measured candidate is missing its surface");
|
|
4794
|
-
const { surface, campaign } = measured;
|
|
4795
|
-
if (!surfaceHashMatches(surface, c.surfaceHash)) throw new Error("buildLoopProvenanceRecord: candidate surface hash does not match its surface bytes");
|
|
4796
|
-
if (campaign.splitDigest !== args.baselineSearchCampaign.splitDigest) throw new Error("buildLoopProvenanceRecord: candidate campaign does not match the search split");
|
|
4797
|
-
const entry = {
|
|
4798
|
-
generation: gen.generationIndex,
|
|
4799
|
-
surfaceHash: c.surfaceHash,
|
|
4800
|
-
contentHash: surfaceContentHash(surface),
|
|
4801
|
-
campaignDigest: campaignMeasurementDigest(campaign),
|
|
4802
|
-
parentSurfaceHash: c.parentSurfaceHash,
|
|
4803
|
-
parentComposite: c.parentComposite,
|
|
4804
|
-
eligibleForPromotion: c.eligibleForPromotion,
|
|
4805
|
-
coverage: {
|
|
4806
|
-
expectedCells: c.coverage.expectedCells,
|
|
4807
|
-
scorableCells: c.coverage.scorableCells,
|
|
4808
|
-
unscorableCells: c.coverage.unscorableCells.map((cell) => ({ ...cell }))
|
|
4809
|
-
},
|
|
4810
|
-
composite: c.composite,
|
|
4811
|
-
promoted: promotedSet.has(c.surfaceHash)
|
|
4812
|
-
};
|
|
4813
|
-
if (c.label) entry.label = c.label;
|
|
4814
|
-
if (c.rationale) entry.rationale = c.rationale;
|
|
4815
|
-
if (c.attribution) entry.attribution = c.attribution;
|
|
4816
|
-
if (c.observedDeltaFromParent !== void 0) entry.observedDeltaFromParent = c.observedDeltaFromParent;
|
|
4817
|
-
candidates.push(entry);
|
|
4818
|
-
}
|
|
4819
|
-
const promotedHash = gen.promoted[0];
|
|
4820
|
-
if (promotedHash) {
|
|
4821
|
-
const promoted = candidateByHash.get(promotedHash);
|
|
4822
|
-
incumbentSurfaceHash = promoted.surfaceHash;
|
|
4823
|
-
if (promoted.composite === null) throw new Error("buildLoopProvenanceRecord: promoted candidate is missing a composite");
|
|
4824
|
-
incumbentComposite = promoted.composite;
|
|
4825
|
-
}
|
|
4826
|
-
}
|
|
4827
|
-
if (surfaceHash(args.winnerSurface) !== incumbentSurfaceHash) throw new Error("buildLoopProvenanceRecord: winner surface does not match the final promoted incumbent");
|
|
4828
|
-
const holdoutDeferred = args.holdout === "deferred";
|
|
4829
|
-
if (args.baselineOnHoldout.splitDigest !== args.winnerOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: baseline and winner use different holdout splits");
|
|
4830
|
-
if (args.neutralizedSurface === void 0 !== (args.neutralizedOnHoldout === void 0)) throw new Error("buildLoopProvenanceRecord: neutralized surface and campaign must be supplied together");
|
|
4831
|
-
if (args.neutralizedOnHoldout && args.neutralizedOnHoldout.splitDigest !== args.baselineOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: neutralized campaign uses a different holdout split");
|
|
4832
|
-
if (holdoutDeferred && args.neutralizedOnHoldout) throw new Error("buildLoopProvenanceRecord: a deferred holdout cannot include a neutralized measurement");
|
|
4833
|
-
const holdoutMeasurement = holdoutDeferred ? { kind: "deferred" } : {
|
|
4834
|
-
kind: "measured",
|
|
4835
|
-
baseline: meanHoldoutComposite(args.baselineOnHoldout),
|
|
4836
|
-
winner: meanHoldoutComposite(args.winnerOnHoldout),
|
|
4837
|
-
...args.neutralizedOnHoldout ? { neutralized: meanHoldoutComposite(args.neutralizedOnHoldout) } : {}
|
|
4838
|
-
};
|
|
4839
|
-
const diff = surfaceContentHash(args.baselineSurface) === surfaceContentHash(args.winnerSurface) ? "" : renderSurfaceDiff(args.winnerSurface, args.baselineSurface);
|
|
4840
|
-
const recordWithoutDigest = {
|
|
4841
|
-
schema: "tangle.loop-provenance",
|
|
4842
|
-
runId: args.runId,
|
|
4843
|
-
runDir: args.runDir,
|
|
4844
|
-
timestamp: args.timestamp,
|
|
4845
|
-
baselineContentHash: surfaceContentHash(args.baselineSurface),
|
|
4846
|
-
winnerContentHash: surfaceContentHash(args.winnerSurface),
|
|
4847
|
-
diff,
|
|
4848
|
-
candidates,
|
|
4849
|
-
evidence: {
|
|
4850
|
-
search: {
|
|
4851
|
-
splitDigest: args.baselineSearchCampaign.splitDigest,
|
|
4852
|
-
baselineCampaignDigest: campaignMeasurementDigest(args.baselineSearchCampaign)
|
|
4853
|
-
},
|
|
4854
|
-
holdout: {
|
|
4855
|
-
splitDigest: args.baselineOnHoldout.splitDigest,
|
|
4856
|
-
baselineCampaignDigest: campaignMeasurementDigest(args.baselineOnHoldout),
|
|
4857
|
-
winnerCampaignDigest: campaignMeasurementDigest(args.winnerOnHoldout),
|
|
4858
|
-
...args.neutralizedSurface && args.neutralizedOnHoldout && holdoutMeasurement.kind === "measured" && holdoutMeasurement.neutralized !== void 0 ? { neutralized: {
|
|
4859
|
-
contentHash: surfaceContentHash(args.neutralizedSurface),
|
|
4860
|
-
campaignDigest: campaignMeasurementDigest(args.neutralizedOnHoldout),
|
|
4861
|
-
composite: holdoutMeasurement.neutralized,
|
|
4862
|
-
lift: holdoutMeasurement.neutralized - holdoutMeasurement.baseline
|
|
4863
|
-
} } : {}
|
|
4864
|
-
},
|
|
4865
|
-
costReceiptsDigest: canonicalDigest([...args.costReceipts].sort((left, right) => compareCodeUnits(left.callId, right.callId)))
|
|
4866
|
-
},
|
|
4867
|
-
baselineSearchComposite,
|
|
4868
|
-
gate: {
|
|
4869
|
-
decision: args.gate.decision,
|
|
4870
|
-
reasons: args.gate.reasons,
|
|
4871
|
-
...args.gate.delta === void 0 ? {} : { delta: args.gate.delta },
|
|
4872
|
-
contributingGates: args.gate.contributingGates.map((g) => ({
|
|
4873
|
-
name: g.name,
|
|
4874
|
-
status: g.status,
|
|
4875
|
-
detail: durableGateDetail(g.detail)
|
|
4876
|
-
}))
|
|
4877
|
-
},
|
|
4878
|
-
...holdoutMeasurement.kind === "deferred" ? { holdout: "deferred" } : {
|
|
4879
|
-
baselineHoldoutComposite: holdoutMeasurement.baseline,
|
|
4880
|
-
winnerHoldoutComposite: holdoutMeasurement.winner,
|
|
4881
|
-
heldOutLift: holdoutMeasurement.winner - holdoutMeasurement.baseline
|
|
4882
|
-
},
|
|
4883
|
-
backend: {
|
|
4884
|
-
verdict: integrity.verdict,
|
|
4885
|
-
workerCallCount: integrity.totalRecords,
|
|
4886
|
-
models,
|
|
4887
|
-
totalInputTokens: integrity.totalInputTokens,
|
|
4888
|
-
totalOutputTokens: integrity.totalOutputTokens,
|
|
4889
|
-
totalCostUsd: integrity.totalCostUsd
|
|
4890
|
-
},
|
|
4891
|
-
totalCostUsd: args.totalCostUsd,
|
|
4892
|
-
totalDurationMs: args.totalDurationMs
|
|
4893
|
-
};
|
|
4894
|
-
if (args.optimizationMethod) recordWithoutDigest.optimizationMethod = durableOptimizationMethod(args.optimizationMethod);
|
|
4895
|
-
if (args.winnerLabel) recordWithoutDigest.winnerLabel = args.winnerLabel;
|
|
4896
|
-
if (args.winnerRationale) recordWithoutDigest.winnerRationale = args.winnerRationale;
|
|
4897
|
-
return {
|
|
4898
|
-
...recordWithoutDigest,
|
|
4899
|
-
recordDigest: canonicalDigest(recordWithoutDigest)
|
|
4900
|
-
};
|
|
4901
|
-
}
|
|
4902
|
-
function durableOptimizationMethod(value) {
|
|
4903
|
-
if (!value || typeof value !== "object" || typeof value.name !== "string" || !value.name.trim() || value.name.trim() !== value.name) throw new Error("buildLoopProvenanceRecord: optimization method name is invalid");
|
|
4904
|
-
if (!value.cost || !Number.isFinite(value.cost.totalCostUsd) || value.cost.totalCostUsd < 0 || typeof value.cost.accountingComplete !== "boolean" || !Array.isArray(value.cost.incompleteReasons) || value.cost.incompleteReasons.some((reason) => typeof reason !== "string" || !reason.trim()) || value.cost.accountingComplete !== (value.cost.incompleteReasons.length === 0)) throw new Error("buildLoopProvenanceRecord: optimization method cost is invalid");
|
|
4905
|
-
if (value.durationMs !== void 0 && (!Number.isFinite(value.durationMs) || value.durationMs < 0)) throw new Error("buildLoopProvenanceRecord: optimization method duration is invalid");
|
|
4906
|
-
try {
|
|
4907
|
-
return JSON.parse(canonicalString(value));
|
|
4908
|
-
} catch (cause) {
|
|
4909
|
-
throw new Error("buildLoopProvenanceRecord: optimization method data must be canonical JSON", { cause });
|
|
4910
|
-
}
|
|
4911
|
-
}
|
|
4912
|
-
/** Digest the exact campaign fields that can affect a measured comparison. */
|
|
4913
|
-
function campaignMeasurementDigest(campaign) {
|
|
4914
|
-
assertCampaignSplitIdentity(campaign.scenarios, campaign.reps, campaign.splitDigest);
|
|
4915
|
-
return canonicalDigest({
|
|
4916
|
-
schema: "tangle.campaign-measurement",
|
|
4917
|
-
manifestHash: campaign.manifestHash,
|
|
4918
|
-
splitDigest: campaign.splitDigest,
|
|
4919
|
-
seed: campaign.seed,
|
|
4920
|
-
reps: campaign.reps,
|
|
4921
|
-
runDir: campaign.runDir,
|
|
4922
|
-
scenarios: campaign.scenarios,
|
|
4923
|
-
cells: [...campaign.cells].sort((left, right) => compareCodeUnits(left.cellId, right.cellId)).map((cell) => ({
|
|
4924
|
-
manifestHash: cell.manifestHash ?? null,
|
|
4925
|
-
cellId: cell.cellId,
|
|
4926
|
-
scenarioId: cell.scenarioId,
|
|
4927
|
-
rep: cell.rep,
|
|
4928
|
-
generation: cell.generation ?? null,
|
|
4929
|
-
judgeScores: cell.judgeScores,
|
|
4930
|
-
costUsd: cell.costUsd,
|
|
4931
|
-
costProvenance: cell.costProvenance,
|
|
4932
|
-
costCallIds: [...cell.costCallIds ?? []].sort(),
|
|
4933
|
-
tokenUsage: cell.tokenUsage,
|
|
4934
|
-
resolvedModels: [...cell.resolvedModels ?? []].sort(),
|
|
4935
|
-
resolvedModel: cell.resolvedModel ?? null,
|
|
4936
|
-
durationMs: cell.durationMs,
|
|
4937
|
-
seed: cell.seed,
|
|
4938
|
-
cached: cell.cached,
|
|
4939
|
-
errorStage: cell.errorStage ?? null,
|
|
4940
|
-
errorJudge: cell.errorJudge ?? null,
|
|
4941
|
-
error: cell.error ?? null
|
|
4942
|
-
}))
|
|
4943
|
-
});
|
|
4944
|
-
}
|
|
4945
|
-
/** Recompute and validate the self-addressed durable record. */
|
|
4946
|
-
function verifyLoopProvenanceRecord(record) {
|
|
4947
|
-
if (record.schema !== "tangle.loop-provenance") throw new Error("loop provenance has an unsupported schema");
|
|
4948
|
-
const { recordDigest, ...recordWithoutDigest } = record;
|
|
4949
|
-
if (recordDigest !== canonicalDigest(recordWithoutDigest)) throw new Error("loop provenance record digest does not match its contents");
|
|
4950
|
-
assertGateContributions(record.gate?.contributingGates, "loop provenance");
|
|
4951
|
-
return record;
|
|
4952
|
-
}
|
|
4953
|
-
/** SHA-256 over the RFC 8785 canonical JSON of `value`. Throws
|
|
4954
|
-
* `LedgerCanonicalizationError` for a value with no canonical form. */
|
|
4955
|
-
function canonicalDigest(value) {
|
|
4956
|
-
return hashCanonical(value);
|
|
4957
|
-
}
|
|
4958
|
-
function durableGateDetail(detail) {
|
|
4959
|
-
if (detail === void 0) return null;
|
|
4960
|
-
try {
|
|
4961
|
-
return JSON.parse(canonicalString(detail));
|
|
4962
|
-
} catch (cause) {
|
|
4963
|
-
throw new Error("buildLoopProvenanceRecord: gate detail must be canonical JSON", { cause });
|
|
4964
|
-
}
|
|
4965
|
-
}
|
|
4966
|
-
function assertGateContributions(value, source) {
|
|
4967
|
-
if (!Array.isArray(value)) throw new Error(`${source}: gate contributingGates must be an array`);
|
|
4968
|
-
const statuses = /* @__PURE__ */ new Set([
|
|
4969
|
-
"pass",
|
|
4970
|
-
"fail",
|
|
4971
|
-
"not_evaluated"
|
|
4972
|
-
]);
|
|
4973
|
-
for (const [index, contribution] of value.entries()) {
|
|
4974
|
-
if (!contribution || typeof contribution !== "object") throw new Error(`${source}: gate contribution ${index} must be an object`);
|
|
4975
|
-
const item = contribution;
|
|
4976
|
-
if (typeof item.name !== "string" || item.name.length === 0) throw new Error(`${source}: gate contribution ${index} must have a non-empty name`);
|
|
4977
|
-
if (!statuses.has(String(item.status))) throw new Error(`${source}: gate contribution '${item.name}' must have status pass, fail, or not_evaluated`);
|
|
4978
|
-
if ("passed" in item) throw new Error(`${source}: gate contribution '${item.name}' uses obsolete passed; use status instead`);
|
|
4979
|
-
}
|
|
4980
|
-
}
|
|
4981
|
-
function validateCandidateMeasurement(candidate, expectedParentHash, expectedParentComposite, promoted) {
|
|
4982
|
-
if (!candidate.parentSurfaceHash || !/^[a-f0-9]{16}$/.test(candidate.parentSurfaceHash)) throw new Error("buildLoopProvenanceRecord: parentSurfaceHash must be 16 lowercase hex characters");
|
|
4983
|
-
if (candidate.parentSurfaceHash !== expectedParentHash) throw new Error("buildLoopProvenanceRecord: candidate parent does not match the incumbent");
|
|
4984
|
-
if (candidate.parentComposite === void 0 || !Number.isFinite(candidate.parentComposite) || Math.abs(candidate.parentComposite - expectedParentComposite) > 1e-12) throw new Error("buildLoopProvenanceRecord: candidate parentComposite does not match the incumbent");
|
|
4985
|
-
if (candidate.observedDeltaFromParent !== void 0) {
|
|
4986
|
-
if (!Number.isFinite(candidate.observedDeltaFromParent)) throw new Error("buildLoopProvenanceRecord: observedDeltaFromParent must be finite");
|
|
4987
|
-
if (candidate.eligibleForPromotion !== true) throw new Error("buildLoopProvenanceRecord: observedDeltaFromParent requires a complete eligible candidate and parentSurfaceHash");
|
|
4988
|
-
}
|
|
4989
|
-
const coverage = candidate.coverage;
|
|
4990
|
-
if (!Number.isSafeInteger(coverage.expectedCells) || coverage.expectedCells <= 0 || !Number.isSafeInteger(coverage.scorableCells) || coverage.scorableCells < 0 || coverage.scorableCells > coverage.expectedCells) throw new Error("buildLoopProvenanceRecord: invalid candidate coverage denominator");
|
|
4991
|
-
const unscorableIds = /* @__PURE__ */ new Set();
|
|
4992
|
-
for (const failure of coverage.unscorableCells) {
|
|
4993
|
-
if (typeof failure.cellId !== "string" || failure.cellId.length === 0 || typeof failure.reason !== "string" || failure.reason.length === 0 || unscorableIds.has(failure.cellId)) throw new Error("buildLoopProvenanceRecord: invalid candidate coverage failures");
|
|
4994
|
-
unscorableIds.add(failure.cellId);
|
|
4995
|
-
}
|
|
4996
|
-
if (coverage.expectedCells - coverage.scorableCells !== coverage.unscorableCells.length) throw new Error("buildLoopProvenanceRecord: candidate coverage counts do not match its failures");
|
|
4997
|
-
const complete = coverage.scorableCells === coverage.expectedCells && coverage.unscorableCells.length === 0;
|
|
4998
|
-
if (candidate.eligibleForPromotion !== complete) throw new Error("buildLoopProvenanceRecord: candidate eligibility contradicts its coverage receipt");
|
|
4999
|
-
if (complete) {
|
|
5000
|
-
if (candidate.composite === null || !Number.isFinite(candidate.composite)) throw new Error("buildLoopProvenanceRecord: complete candidate composite must be finite");
|
|
5001
|
-
if (candidate.observedDeltaFromParent === void 0) throw new Error("buildLoopProvenanceRecord: complete candidate is missing observedDeltaFromParent");
|
|
5002
|
-
const recomputed = candidate.composite - candidate.parentComposite;
|
|
5003
|
-
if (Math.abs(candidate.observedDeltaFromParent - recomputed) > 1e-12) throw new Error("buildLoopProvenanceRecord: observed delta does not match measured scores");
|
|
5004
|
-
} else {
|
|
5005
|
-
if (candidate.composite !== null && !Number.isFinite(candidate.composite)) throw new Error("buildLoopProvenanceRecord: candidate composite must be finite or null");
|
|
5006
|
-
if (candidate.observedDeltaFromParent !== void 0) throw new Error("buildLoopProvenanceRecord: incomplete candidate cannot carry observed delta");
|
|
5007
|
-
}
|
|
5008
|
-
if (promoted && (!complete || (candidate.observedDeltaFromParent ?? 0) <= 0)) throw new Error("buildLoopProvenanceRecord: promoted candidate must improve the incumbent");
|
|
5009
|
-
}
|
|
5010
|
-
function hashId(parts) {
|
|
5011
|
-
return createHash("sha256").update(parts.join(":")).digest("hex");
|
|
5012
|
-
}
|
|
5013
|
-
/**
|
|
5014
|
-
* Build the loop's OTLP-ingestable spans from a provenance record. One root
|
|
5015
|
-
* span per loop (`tangle.runId`), one span per generation, one span per
|
|
5016
|
-
* candidate (carrying its surfaceHash + label), and one span for the gate
|
|
5017
|
-
* decision (carrying reasons + delta + lift). Candidate + gate spans pivot on
|
|
5018
|
-
* the same `tangle.runId` / `tangle.generation` attributes `/adapters/otel`
|
|
5019
|
-
* reads, so the hosted collector reconstructs the full tree.
|
|
5020
|
-
*
|
|
5021
|
-
* Times are synthesized monotonically off a single base so the span tree is
|
|
5022
|
-
* orderable; the substrate does not retain per-candidate wall-clock starts.
|
|
5023
|
-
*/
|
|
5024
|
-
function loopProvenanceSpans(record, opts = {}) {
|
|
5025
|
-
const traceId = hashId(["trace", record.runId]).slice(0, 32);
|
|
5026
|
-
const baseTimeMs = opts.baseTimeMs ?? (Date.parse(record.timestamp) || Date.now());
|
|
5027
|
-
const durationMs = Math.max(1, record.totalDurationMs);
|
|
5028
|
-
if (!Number.isSafeInteger(baseTimeMs) || baseTimeMs < 0) throw new RangeError("loop provenance baseTimeMs must be a non-negative safe integer");
|
|
5029
|
-
if (!Number.isSafeInteger(durationMs)) throw new RangeError("loop provenance duration must be a safe integer number of milliseconds");
|
|
5030
|
-
const baseTime = BigInt(baseTimeMs);
|
|
5031
|
-
const baseNano = (baseTime * 1000000n).toString();
|
|
5032
|
-
const endNano = ((baseTime + BigInt(durationMs)) * 1000000n).toString();
|
|
5033
|
-
const spans = [];
|
|
5034
|
-
const rootSpanId = hashId(["root", record.runId]).slice(0, 16);
|
|
5035
|
-
const rootAttributes = {
|
|
5036
|
-
"tangle.runId": record.runId,
|
|
5037
|
-
"tangle.runDir": record.runDir,
|
|
5038
|
-
"tangle.baselineContentHash": record.baselineContentHash,
|
|
5039
|
-
"tangle.winnerContentHash": record.winnerContentHash,
|
|
5040
|
-
"tangle.baselineSearchComposite": record.baselineSearchComposite,
|
|
5041
|
-
"tangle.gateDecision": record.gate.decision,
|
|
5042
|
-
"tangle.backendVerdict": record.backend.verdict,
|
|
5043
|
-
"tangle.workerCallCount": record.backend.workerCallCount,
|
|
5044
|
-
"tangle.totalCostUsd": record.totalCostUsd
|
|
5045
|
-
};
|
|
5046
|
-
if (record.heldOutLift !== void 0) rootAttributes["tangle.heldOutLift"] = record.heldOutLift;
|
|
5047
|
-
if (record.holdout) rootAttributes["tangle.holdout"] = record.holdout;
|
|
5048
|
-
spans.push({
|
|
5049
|
-
traceId,
|
|
5050
|
-
spanId: rootSpanId,
|
|
5051
|
-
name: "improvement-loop",
|
|
5052
|
-
startTimeUnixNano: baseNano,
|
|
5053
|
-
endTimeUnixNano: endNano,
|
|
5054
|
-
attributes: rootAttributes,
|
|
5055
|
-
status: { code: "OK" },
|
|
5056
|
-
"tangle.runId": record.runId
|
|
5057
|
-
});
|
|
5058
|
-
const byGen = /* @__PURE__ */ new Map();
|
|
5059
|
-
for (const c of record.candidates) {
|
|
5060
|
-
const arr = byGen.get(c.generation) ?? [];
|
|
5061
|
-
arr.push(c);
|
|
5062
|
-
byGen.set(c.generation, arr);
|
|
5063
|
-
}
|
|
5064
|
-
for (const [generation, cands] of [...byGen.entries()].sort((a, b) => a[0] - b[0])) {
|
|
5065
|
-
const genSpanId = hashId([
|
|
5066
|
-
"gen",
|
|
5067
|
-
record.runId,
|
|
5068
|
-
String(generation)
|
|
5069
|
-
]).slice(0, 16);
|
|
5070
|
-
const measuredComposites = cands.flatMap((candidate) => candidate.composite === null ? [] : [candidate.composite]);
|
|
5071
|
-
spans.push({
|
|
5072
|
-
traceId,
|
|
5073
|
-
spanId: genSpanId,
|
|
5074
|
-
parentSpanId: rootSpanId,
|
|
5075
|
-
name: `generation-${generation}`,
|
|
5076
|
-
startTimeUnixNano: baseNano,
|
|
5077
|
-
endTimeUnixNano: endNano,
|
|
5078
|
-
attributes: {
|
|
5079
|
-
"tangle.runId": record.runId,
|
|
5080
|
-
"tangle.generation": generation,
|
|
5081
|
-
"tangle.populationSize": cands.length,
|
|
5082
|
-
...measuredComposites.length > 0 ? { "tangle.bestComposite": Math.max(...measuredComposites) } : {}
|
|
5083
|
-
},
|
|
5084
|
-
"tangle.runId": record.runId,
|
|
5085
|
-
"tangle.generation": generation
|
|
5086
|
-
});
|
|
5087
|
-
for (let i = 0; i < cands.length; i++) {
|
|
5088
|
-
const c = cands[i];
|
|
5089
|
-
const candSpanId = hashId([
|
|
5090
|
-
"cand",
|
|
5091
|
-
record.runId,
|
|
5092
|
-
String(generation),
|
|
5093
|
-
c.surfaceHash
|
|
5094
|
-
]).slice(0, 16);
|
|
5095
|
-
const attributes = {
|
|
5096
|
-
"tangle.runId": record.runId,
|
|
5097
|
-
"tangle.generation": generation,
|
|
5098
|
-
"tangle.surfaceHash": c.surfaceHash,
|
|
5099
|
-
"tangle.contentHash": c.contentHash,
|
|
5100
|
-
"tangle.parentSurfaceHash": c.parentSurfaceHash,
|
|
5101
|
-
"tangle.parentComposite": c.parentComposite,
|
|
5102
|
-
"tangle.eligibleForPromotion": c.eligibleForPromotion,
|
|
5103
|
-
"tangle.expectedCells": c.coverage.expectedCells,
|
|
5104
|
-
"tangle.scorableCells": c.coverage.scorableCells,
|
|
5105
|
-
"tangle.unscorableCells": c.coverage.unscorableCells.length,
|
|
5106
|
-
"tangle.promoted": c.promoted
|
|
5107
|
-
};
|
|
5108
|
-
if (c.composite !== null) attributes["tangle.composite"] = c.composite;
|
|
5109
|
-
if (c.observedDeltaFromParent !== void 0) attributes["tangle.observedDeltaFromParent"] = c.observedDeltaFromParent;
|
|
5110
|
-
if (c.label) attributes["tangle.candidateLabel"] = c.label;
|
|
5111
|
-
if (c.rationale) attributes["tangle.candidateRationale"] = c.rationale;
|
|
5112
|
-
spans.push({
|
|
5113
|
-
traceId,
|
|
5114
|
-
spanId: candSpanId,
|
|
5115
|
-
parentSpanId: genSpanId,
|
|
5116
|
-
name: `candidate-${c.surfaceHash}`,
|
|
5117
|
-
startTimeUnixNano: baseNano,
|
|
5118
|
-
endTimeUnixNano: endNano,
|
|
5119
|
-
attributes,
|
|
5120
|
-
"tangle.runId": record.runId,
|
|
5121
|
-
"tangle.generation": generation
|
|
5122
|
-
});
|
|
5123
|
-
}
|
|
5124
|
-
}
|
|
5125
|
-
const gateSpanId = hashId(["gate", record.runId]).slice(0, 16);
|
|
5126
|
-
const gateAttributes = {
|
|
5127
|
-
"tangle.runId": record.runId,
|
|
5128
|
-
"tangle.gateDecision": record.gate.decision,
|
|
5129
|
-
"tangle.gateReasons": JSON.stringify(record.gate.reasons)
|
|
5130
|
-
};
|
|
5131
|
-
const gateDelta = record.gate.delta ?? record.heldOutLift;
|
|
5132
|
-
if (gateDelta !== void 0) gateAttributes["tangle.gateDelta"] = gateDelta;
|
|
5133
|
-
if (record.heldOutLift !== void 0) gateAttributes["tangle.heldOutLift"] = record.heldOutLift;
|
|
5134
|
-
if (record.baselineHoldoutComposite !== void 0) gateAttributes["tangle.baselineHoldoutComposite"] = record.baselineHoldoutComposite;
|
|
5135
|
-
if (record.winnerHoldoutComposite !== void 0) gateAttributes["tangle.winnerHoldoutComposite"] = record.winnerHoldoutComposite;
|
|
5136
|
-
if (record.holdout) gateAttributes["tangle.holdout"] = record.holdout;
|
|
5137
|
-
spans.push({
|
|
5138
|
-
traceId,
|
|
5139
|
-
spanId: gateSpanId,
|
|
5140
|
-
parentSpanId: rootSpanId,
|
|
5141
|
-
name: "gate-decision",
|
|
5142
|
-
startTimeUnixNano: endNano,
|
|
5143
|
-
endTimeUnixNano: endNano,
|
|
5144
|
-
attributes: gateAttributes,
|
|
5145
|
-
status: { code: "OK" },
|
|
5146
|
-
"tangle.runId": record.runId
|
|
5147
|
-
});
|
|
5148
|
-
return spans;
|
|
5149
|
-
}
|
|
5150
|
-
/** Canonical durable paths under the run dir. */
|
|
5151
|
-
function provenanceRecordPath(runDir) {
|
|
5152
|
-
return join(runDir, "loop-provenance.json");
|
|
5153
|
-
}
|
|
5154
|
-
/**
|
|
5155
|
-
* Canonical path for the durable OTLP spans JSONL file under a loop run directory.
|
|
5156
|
-
*/
|
|
5157
|
-
function provenanceSpansPath(runDir) {
|
|
5158
|
-
return join(runDir, "loop-provenance-spans.jsonl");
|
|
5159
|
-
}
|
|
5160
|
-
/** Snapshot a held-out campaign into the hosted `EvalRunGenerationSnapshot`
|
|
5161
|
-
* shape — per-cell composite + per-judge dimensions, aggregate mean, cost,
|
|
5162
|
-
* duration. The dashboard renders these as the baseline → winner comparison. */
|
|
5163
|
-
function snapshotFromHoldout(index, surfaceHash, surface, campaign) {
|
|
5164
|
-
return {
|
|
5165
|
-
index,
|
|
5166
|
-
surfaceHash,
|
|
5167
|
-
surface,
|
|
5168
|
-
cells: campaign.cells.map((cell) => {
|
|
5169
|
-
const execution = campaignCellExecutionEvidence(cell);
|
|
5170
|
-
const quality = projectCampaignCellQuality(cell);
|
|
5171
|
-
const score = {
|
|
5172
|
-
scenarioId: cell.scenarioId,
|
|
5173
|
-
rep: cell.rep,
|
|
5174
|
-
compositeMean: quality.score ?? null,
|
|
5175
|
-
dimensions: quality.judgeScores?.perJudge ?? {},
|
|
5176
|
-
terminalOutcome: execution.terminalOutcome,
|
|
5177
|
-
executionErrorCount: execution.executionErrorCount ?? null
|
|
5178
|
-
};
|
|
5179
|
-
if (cell.error) score.errorMessage = cell.error;
|
|
5180
|
-
return score;
|
|
5181
|
-
}),
|
|
5182
|
-
compositeMean: campaignMeanCompositeOrNull(campaign),
|
|
5183
|
-
costUsd: campaign.aggregates.cost.totalCostUsd,
|
|
5184
|
-
durationMs: campaign.durationMs
|
|
5185
|
-
};
|
|
5186
|
-
}
|
|
5187
|
-
/** Build the hosted `EvalRunEvent` from the loop args + record — baseline +
|
|
5188
|
-
* winner snapshots, gate decision, held-out lift, cost, duration. Shipped to
|
|
5189
|
-
* `/v1/ingest/eval-runs` so the run appears in the dashboard's run list (the
|
|
5190
|
-
* trace spans, shipped separately, back the per-candidate drill-down). */
|
|
5191
|
-
function buildEvalRunEvent(args, record) {
|
|
5192
|
-
return {
|
|
5193
|
-
runId: args.runId,
|
|
5194
|
-
runDir: args.runDir,
|
|
5195
|
-
timestamp: args.timestamp,
|
|
5196
|
-
status: "finished",
|
|
5197
|
-
labels: {},
|
|
5198
|
-
baseline: snapshotFromHoldout(0, record.baselineContentHash, args.baselineSurface, args.baselineOnHoldout),
|
|
5199
|
-
generations: [snapshotFromHoldout(1, record.winnerContentHash, args.winnerSurface, args.winnerOnHoldout)],
|
|
5200
|
-
gateDecision: args.gate.decision,
|
|
5201
|
-
...record.heldOutLift !== void 0 ? { holdoutLift: record.heldOutLift } : {},
|
|
5202
|
-
totalCostUsd: args.totalCostUsd,
|
|
5203
|
-
totalDurationMs: args.totalDurationMs
|
|
5204
|
-
};
|
|
5205
|
-
}
|
|
5206
|
-
/**
|
|
5207
|
-
* Build the provenance record + OTel spans and persist them durably under the
|
|
5208
|
-
* run dir (and ship spans to a hosted collector when one is wired). Returns
|
|
5209
|
-
* both artifacts so the caller can assert on / re-derive from them.
|
|
5210
|
-
*
|
|
5211
|
-
* Fail-loud: the durable write throws on storage failure (a swallowed write is
|
|
5212
|
-
* exactly the "emitted but lost" failure this closes). The hosted span ship is
|
|
5213
|
-
* the one best-effort leg — its failure is logged, not thrown, so an offline
|
|
5214
|
-
* collector never fails the loop (the durable artifact is the source of truth).
|
|
5215
|
-
*/
|
|
5216
|
-
async function emitLoopProvenance(args) {
|
|
5217
|
-
const record = buildLoopProvenanceRecord(args);
|
|
5218
|
-
const spans = loopProvenanceSpans(record);
|
|
5219
|
-
args.storage.ensureDir(args.runDir);
|
|
5220
|
-
const recordPath = provenanceRecordPath(args.runDir);
|
|
5221
|
-
const spansPath = provenanceSpansPath(args.runDir);
|
|
5222
|
-
args.storage.write(recordPath, JSON.stringify(record, null, 2));
|
|
5223
|
-
args.storage.write(spansPath, spans.map((s) => JSON.stringify(s)).join("\n"));
|
|
5224
|
-
if (args.hostedClient) {
|
|
5225
|
-
try {
|
|
5226
|
-
await args.hostedClient.ingestEvalRun(buildEvalRunEvent(args, record));
|
|
5227
|
-
} catch (err) {
|
|
5228
|
-
const msg = err instanceof Error ? err.message : String(err);
|
|
5229
|
-
console.warn(`[agent-eval] hosted eval-run ingest failed (continuing): ${msg}`);
|
|
5230
|
-
}
|
|
5231
|
-
try {
|
|
5232
|
-
await args.hostedClient.ingestTraces(spans);
|
|
5233
|
-
} catch (err) {
|
|
5234
|
-
const msg = err instanceof Error ? err.message : String(err);
|
|
5235
|
-
console.warn(`[agent-eval] provenance span ingest failed (continuing): ${msg}`);
|
|
5236
|
-
}
|
|
5237
|
-
}
|
|
5238
|
-
return {
|
|
5239
|
-
record,
|
|
5240
|
-
spans,
|
|
5241
|
-
recordPath,
|
|
5242
|
-
spansPath
|
|
5243
|
-
};
|
|
5244
|
-
}
|
|
5245
|
-
//#endregion
|
|
5246
|
-
//#region src/campaign/optimization-cost.ts
|
|
5247
|
-
/** Attribute method calls while retaining the shared account's admission and read behavior. */
|
|
5248
|
-
function createMethodCostScope(account, methodName) {
|
|
5249
|
-
const tags = { optimizationAttempt: crypto.randomUUID() };
|
|
5250
|
-
return {
|
|
5251
|
-
ledger: Object.freeze({
|
|
5252
|
-
costCeilingUsd: account.costCeilingUsd,
|
|
5253
|
-
runPaidCall: (input) => account.runPaidCall({
|
|
5254
|
-
...input,
|
|
5255
|
-
tags: {
|
|
5256
|
-
...input.tags,
|
|
5257
|
-
...tags
|
|
5258
|
-
}
|
|
5259
|
-
}),
|
|
5260
|
-
summary: account.summary.bind(account),
|
|
5261
|
-
list: account.list.bind(account),
|
|
5262
|
-
reconcile: account.reconcile.bind(account),
|
|
5263
|
-
markCompleted: account.markCompleted.bind(account),
|
|
5264
|
-
costPerCompletedTask: account.costPerCompletedTask.bind(account),
|
|
5265
|
-
...account.listPending ? { listPending: account.listPending.bind(account) } : {},
|
|
5266
|
-
...account.waitForIdle ? { waitForIdle: account.waitForIdle.bind(account) } : {}
|
|
5267
|
-
}),
|
|
5268
|
-
reconcile(reported) {
|
|
5269
|
-
const summary = account.summary({ tags });
|
|
5270
|
-
if (summary.pendingCalls > 0) throw new Error(`optimization method '${methodName}' returned with ${summary.pendingCalls} pending paid call(s)`);
|
|
5271
|
-
const recorded = costFromLedgerSummary(summary);
|
|
5272
|
-
const totalCostUsd = Math.max(reported.totalCostUsd, recorded.totalCostUsd);
|
|
5273
|
-
const roundingToleranceUsd = Number.EPSILON * Math.max(1, totalCostUsd) * (summary.totalCalls + 1);
|
|
5274
|
-
const combined = combineComparisonCosts([{
|
|
5275
|
-
label: "reported",
|
|
5276
|
-
cost: reported
|
|
5277
|
-
}, {
|
|
5278
|
-
label: "recorded",
|
|
5279
|
-
cost: recorded
|
|
5280
|
-
}]);
|
|
5281
|
-
const incompleteReasons = [
|
|
5282
|
-
...reported.incompleteReasons,
|
|
5283
|
-
...recorded.incompleteReasons.map((reason) => `recorded: ${reason}`),
|
|
5284
|
-
...recorded.totalCostUsd - reported.totalCostUsd > roundingToleranceUsd ? [`reported ${reported.totalCostUsd} USD below recorded ${recorded.totalCostUsd} USD`] : []
|
|
5285
|
-
];
|
|
5286
|
-
return {
|
|
5287
|
-
totalCostUsd,
|
|
5288
|
-
costProvenance: combined.costProvenance.kind === "uncaptured" ? combined.costProvenance : {
|
|
5289
|
-
kind: combined.costProvenance.kind,
|
|
5290
|
-
usd: totalCostUsd
|
|
5291
|
-
},
|
|
5292
|
-
accountingComplete: incompleteReasons.length === 0,
|
|
5293
|
-
incompleteReasons
|
|
5294
|
-
};
|
|
5295
|
-
}
|
|
5296
|
-
};
|
|
5297
|
-
}
|
|
5298
|
-
/** Keep the cost fields a custom optimization method must report. */
|
|
5299
|
-
function costFromLedgerSummary(summary) {
|
|
5300
|
-
const cost = {
|
|
5301
|
-
totalCostUsd: summary.totalCostUsd,
|
|
5302
|
-
costProvenance: structuredClone(summary.costProvenance),
|
|
5303
|
-
accountingComplete: summary.accountingComplete,
|
|
5304
|
-
incompleteReasons: [...summary.incompleteReasons]
|
|
5305
|
-
};
|
|
5306
|
-
assertComparisonCost(cost, "cost ledger");
|
|
5307
|
-
return cost;
|
|
5308
|
-
}
|
|
5309
|
-
/** Combine method costs without turning one unknown bill into a known total. */
|
|
5310
|
-
function combineComparisonCosts(entries) {
|
|
5311
|
-
const totalCostUsd = entries.reduce((total, entry) => total + entry.cost.totalCostUsd, 0);
|
|
5312
|
-
const cost = {
|
|
5313
|
-
totalCostUsd,
|
|
5314
|
-
costProvenance: entries.some((entry) => entry.cost.costProvenance.kind === "uncaptured") ? {
|
|
5315
|
-
kind: "uncaptured",
|
|
5316
|
-
usd: null
|
|
5317
|
-
} : entries.every((entry) => entry.cost.costProvenance.kind === "observed") ? {
|
|
5318
|
-
kind: "observed",
|
|
5319
|
-
usd: totalCostUsd
|
|
5320
|
-
} : {
|
|
5321
|
-
kind: "estimated",
|
|
5322
|
-
usd: totalCostUsd
|
|
5323
|
-
},
|
|
5324
|
-
accountingComplete: entries.every((entry) => entry.cost.accountingComplete),
|
|
5325
|
-
incompleteReasons: entries.flatMap((entry) => entry.cost.incompleteReasons.map((reason) => `${entry.label}: ${reason}`))
|
|
5326
|
-
};
|
|
5327
|
-
assertComparisonCost(cost, "combined cost");
|
|
5328
|
-
return cost;
|
|
5329
|
-
}
|
|
5330
|
-
function assertComparisonCost(cost, label) {
|
|
5331
|
-
if (!cost || typeof cost !== "object") throw new Error(`compareOptimizationMethods: ${label} returned no cost`);
|
|
5332
|
-
if (!Number.isFinite(cost.totalCostUsd) || cost.totalCostUsd < 0) throw new Error(`compareOptimizationMethods: ${label} returned an invalid totalCostUsd`);
|
|
5333
|
-
const provenance = cost.costProvenance;
|
|
5334
|
-
if (!provenance || typeof provenance !== "object" || provenance.kind !== "observed" && provenance.kind !== "estimated" && provenance.kind !== "uncaptured" || (provenance.kind === "uncaptured" ? provenance.usd !== null : !Number.isFinite(provenance.usd) || provenance.usd < 0)) throw new Error(`compareOptimizationMethods: ${label} returned invalid costProvenance`);
|
|
5335
|
-
if (provenance.kind !== "uncaptured" && provenance.usd !== cost.totalCostUsd) throw new Error(`compareOptimizationMethods: ${label} returned costProvenance inconsistent with totalCostUsd`);
|
|
5336
|
-
if (typeof cost.accountingComplete !== "boolean") throw new Error(`compareOptimizationMethods: ${label} returned invalid accountingComplete`);
|
|
5337
|
-
if (!Array.isArray(cost.incompleteReasons) || cost.incompleteReasons.some((reason) => typeof reason !== "string" || reason.trim().length === 0)) throw new Error(`compareOptimizationMethods: ${label} returned invalid incompleteReasons`);
|
|
5338
|
-
if (cost.accountingComplete !== (cost.incompleteReasons.length === 0)) throw new Error(`compareOptimizationMethods: ${label} returned inconsistent cost completeness and reasons`);
|
|
5339
|
-
if (cost.accountingComplete && provenance.kind === "uncaptured") throw new Error(`compareOptimizationMethods: ${label} cannot mark uncaptured cost as complete`);
|
|
5340
|
-
}
|
|
5341
|
-
//#endregion
|
|
5342
|
-
//#region src/campaign/gepa-candidate-population.ts
|
|
5343
|
-
/**
|
|
5344
|
-
* Read GEPA's exact candidate graph from the artifact addressed by method provenance.
|
|
5345
|
-
*
|
|
5346
|
-
* The reader checks the supplied digest, declared byte count, run identity,
|
|
5347
|
-
* candidate surfaces, parent graph, selection scores, and configured bounds.
|
|
5348
|
-
* This proves that the bytes match the supplied summary. The caller remains
|
|
5349
|
-
* responsible for obtaining that summary from trusted method provenance.
|
|
5350
|
-
*/
|
|
5351
|
-
function readGepaCandidatePopulationArtifact(input) {
|
|
5352
|
-
assertGepaCandidatePopulationSummary(input.summary);
|
|
5353
|
-
const scenarioIds = scenarioIdSet(input.summary.scenarioIds);
|
|
5354
|
-
const storage = input.storage ?? fsCampaignStorage();
|
|
5355
|
-
const contents = storage.read(input.summary.path);
|
|
5356
|
-
if (contents === void 0) {
|
|
5357
|
-
const state = storage.exists(input.summary.path) ? "unreadable" : "missing";
|
|
5358
|
-
throw new Error(`GEPA candidate population artifact is ${state} at '${input.summary.path}'`);
|
|
4679
|
+
function readGepaCandidatePopulationArtifact(input) {
|
|
4680
|
+
assertGepaCandidatePopulationSummary(input.summary);
|
|
4681
|
+
const scenarioIds = scenarioIdSet(input.summary.scenarioIds);
|
|
4682
|
+
const storage = input.storage ?? fsCampaignStorage();
|
|
4683
|
+
const contents = storage.read(input.summary.path);
|
|
4684
|
+
if (contents === void 0) {
|
|
4685
|
+
const state = storage.exists(input.summary.path) ? "unreadable" : "missing";
|
|
4686
|
+
throw new Error(`GEPA candidate population artifact is ${state} at '${input.summary.path}'`);
|
|
5359
4687
|
}
|
|
5360
4688
|
const bytes = new TextEncoder().encode(contents).byteLength;
|
|
5361
4689
|
if (bytes !== input.summary.bytes) throw new Error(`GEPA candidate population byte count mismatch at '${input.summary.path}': expected ${input.summary.bytes}, got ${bytes}`);
|
|
@@ -5524,6 +4852,213 @@ function assertPositiveSafeInteger(value, label) {
|
|
|
5524
4852
|
if (!Number.isSafeInteger(value) || value <= 0) throw new Error(`GEPA candidate population ${label} must be a positive safe integer`);
|
|
5525
4853
|
}
|
|
5526
4854
|
//#endregion
|
|
4855
|
+
//#region src/campaign/optimization-cost.ts
|
|
4856
|
+
/** Attribute method calls while retaining the shared account's admission and read behavior. */
|
|
4857
|
+
function createMethodCostScope(account, methodName) {
|
|
4858
|
+
const tags = { optimizationAttempt: crypto.randomUUID() };
|
|
4859
|
+
return {
|
|
4860
|
+
ledger: Object.freeze({
|
|
4861
|
+
costCeilingUsd: account.costCeilingUsd,
|
|
4862
|
+
runPaidCall: (input) => account.runPaidCall({
|
|
4863
|
+
...input,
|
|
4864
|
+
tags: {
|
|
4865
|
+
...input.tags,
|
|
4866
|
+
...tags
|
|
4867
|
+
}
|
|
4868
|
+
}),
|
|
4869
|
+
summary: account.summary.bind(account),
|
|
4870
|
+
list: account.list.bind(account),
|
|
4871
|
+
reconcile: account.reconcile.bind(account),
|
|
4872
|
+
markCompleted: account.markCompleted.bind(account),
|
|
4873
|
+
costPerCompletedTask: account.costPerCompletedTask.bind(account),
|
|
4874
|
+
...account.listPending ? { listPending: account.listPending.bind(account) } : {},
|
|
4875
|
+
...account.waitForIdle ? { waitForIdle: account.waitForIdle.bind(account) } : {}
|
|
4876
|
+
}),
|
|
4877
|
+
reconcile(reported) {
|
|
4878
|
+
const summary = account.summary({ tags });
|
|
4879
|
+
if (summary.pendingCalls > 0) throw new Error(`optimization method '${methodName}' returned with ${summary.pendingCalls} pending paid call(s)`);
|
|
4880
|
+
const recorded = costFromLedgerSummary(summary);
|
|
4881
|
+
const totalCostUsd = Math.max(reported.totalCostUsd, recorded.totalCostUsd);
|
|
4882
|
+
const roundingToleranceUsd = Number.EPSILON * Math.max(1, totalCostUsd) * (summary.totalCalls + 1);
|
|
4883
|
+
const combined = combineComparisonCosts([{
|
|
4884
|
+
label: "reported",
|
|
4885
|
+
cost: reported
|
|
4886
|
+
}, {
|
|
4887
|
+
label: "recorded",
|
|
4888
|
+
cost: recorded
|
|
4889
|
+
}]);
|
|
4890
|
+
const incompleteReasons = [
|
|
4891
|
+
...reported.incompleteReasons,
|
|
4892
|
+
...recorded.incompleteReasons.map((reason) => `recorded: ${reason}`),
|
|
4893
|
+
...recorded.totalCostUsd - reported.totalCostUsd > roundingToleranceUsd ? [`reported ${reported.totalCostUsd} USD below recorded ${recorded.totalCostUsd} USD`] : []
|
|
4894
|
+
];
|
|
4895
|
+
return {
|
|
4896
|
+
totalCostUsd,
|
|
4897
|
+
costProvenance: combined.costProvenance.kind === "uncaptured" ? combined.costProvenance : {
|
|
4898
|
+
kind: combined.costProvenance.kind,
|
|
4899
|
+
usd: totalCostUsd
|
|
4900
|
+
},
|
|
4901
|
+
accountingComplete: incompleteReasons.length === 0,
|
|
4902
|
+
incompleteReasons
|
|
4903
|
+
};
|
|
4904
|
+
}
|
|
4905
|
+
};
|
|
4906
|
+
}
|
|
4907
|
+
/** Keep the cost fields a custom optimization method must report. */
|
|
4908
|
+
function costFromLedgerSummary(summary) {
|
|
4909
|
+
const cost = {
|
|
4910
|
+
totalCostUsd: summary.totalCostUsd,
|
|
4911
|
+
costProvenance: structuredClone(summary.costProvenance),
|
|
4912
|
+
accountingComplete: summary.accountingComplete,
|
|
4913
|
+
incompleteReasons: [...summary.incompleteReasons]
|
|
4914
|
+
};
|
|
4915
|
+
assertComparisonCost(cost, "cost ledger");
|
|
4916
|
+
return cost;
|
|
4917
|
+
}
|
|
4918
|
+
/** Combine method costs without turning one unknown bill into a known total. */
|
|
4919
|
+
function combineComparisonCosts(entries) {
|
|
4920
|
+
const totalCostUsd = entries.reduce((total, entry) => total + entry.cost.totalCostUsd, 0);
|
|
4921
|
+
const cost = {
|
|
4922
|
+
totalCostUsd,
|
|
4923
|
+
costProvenance: entries.some((entry) => entry.cost.costProvenance.kind === "uncaptured") ? {
|
|
4924
|
+
kind: "uncaptured",
|
|
4925
|
+
usd: null
|
|
4926
|
+
} : entries.every((entry) => entry.cost.costProvenance.kind === "observed") ? {
|
|
4927
|
+
kind: "observed",
|
|
4928
|
+
usd: totalCostUsd
|
|
4929
|
+
} : {
|
|
4930
|
+
kind: "estimated",
|
|
4931
|
+
usd: totalCostUsd
|
|
4932
|
+
},
|
|
4933
|
+
accountingComplete: entries.every((entry) => entry.cost.accountingComplete),
|
|
4934
|
+
incompleteReasons: entries.flatMap((entry) => entry.cost.incompleteReasons.map((reason) => `${entry.label}: ${reason}`))
|
|
4935
|
+
};
|
|
4936
|
+
assertComparisonCost(cost, "combined cost");
|
|
4937
|
+
return cost;
|
|
4938
|
+
}
|
|
4939
|
+
function assertComparisonCost(cost, label) {
|
|
4940
|
+
if (!cost || typeof cost !== "object") throw new Error(`compareOptimizationMethods: ${label} returned no cost`);
|
|
4941
|
+
if (!Number.isFinite(cost.totalCostUsd) || cost.totalCostUsd < 0) throw new Error(`compareOptimizationMethods: ${label} returned an invalid totalCostUsd`);
|
|
4942
|
+
const provenance = cost.costProvenance;
|
|
4943
|
+
if (!provenance || typeof provenance !== "object" || provenance.kind !== "observed" && provenance.kind !== "estimated" && provenance.kind !== "uncaptured" || (provenance.kind === "uncaptured" ? provenance.usd !== null : !Number.isFinite(provenance.usd) || provenance.usd < 0)) throw new Error(`compareOptimizationMethods: ${label} returned invalid costProvenance`);
|
|
4944
|
+
if (provenance.kind !== "uncaptured" && provenance.usd !== cost.totalCostUsd) throw new Error(`compareOptimizationMethods: ${label} returned costProvenance inconsistent with totalCostUsd`);
|
|
4945
|
+
if (typeof cost.accountingComplete !== "boolean") throw new Error(`compareOptimizationMethods: ${label} returned invalid accountingComplete`);
|
|
4946
|
+
if (!Array.isArray(cost.incompleteReasons) || cost.incompleteReasons.some((reason) => typeof reason !== "string" || reason.trim().length === 0)) throw new Error(`compareOptimizationMethods: ${label} returned invalid incompleteReasons`);
|
|
4947
|
+
if (cost.accountingComplete !== (cost.incompleteReasons.length === 0)) throw new Error(`compareOptimizationMethods: ${label} returned inconsistent cost completeness and reasons`);
|
|
4948
|
+
if (cost.accountingComplete && provenance.kind === "uncaptured") throw new Error(`compareOptimizationMethods: ${label} cannot mark uncaptured cost as complete`);
|
|
4949
|
+
}
|
|
4950
|
+
//#endregion
|
|
4951
|
+
//#region src/campaign/optimization-method.ts
|
|
4952
|
+
/** Both complete-method workflows use the same detached inputs, accounting, and evidence admission. */
|
|
4953
|
+
async function executeOptimizationMethod(options) {
|
|
4954
|
+
assertSearchHistoryAdmissionOptions(options);
|
|
4955
|
+
const { method, input } = options;
|
|
4956
|
+
const costScope = createMethodCostScope(input.costLedger, method.name);
|
|
4957
|
+
const cloneScenarios = (scenarios) => Object.freeze(scenarios.map((scenario) => structuredClone(scenario)));
|
|
4958
|
+
const selected = structuredClone(await method.optimize(Object.freeze({
|
|
4959
|
+
...input,
|
|
4960
|
+
baselineSurface: structuredClone(input.baselineSurface),
|
|
4961
|
+
trainScenarios: cloneScenarios(input.trainScenarios),
|
|
4962
|
+
selectionScenarios: cloneScenarios(input.selectionScenarios),
|
|
4963
|
+
judges: Object.freeze(input.judges.map((judge) => {
|
|
4964
|
+
const dimensions = judge.dimensions.map((dimension) => Object.freeze({ ...dimension }));
|
|
4965
|
+
Object.freeze(dimensions);
|
|
4966
|
+
return Object.freeze({
|
|
4967
|
+
...judge,
|
|
4968
|
+
dimensions
|
|
4969
|
+
});
|
|
4970
|
+
})),
|
|
4971
|
+
runOptions: Object.freeze({ ...input.runOptions }),
|
|
4972
|
+
costLedger: costScope.ledger
|
|
4973
|
+
})));
|
|
4974
|
+
assertOptimizationResult(method.name, selected);
|
|
4975
|
+
const history = searchHistoryCoverageRow(method.name, selected.searchHistory);
|
|
4976
|
+
if (options.searchHistoryPolicy === "require-complete") assertCompleteSearchHistory(method.name, selected.searchHistory);
|
|
4977
|
+
if (options.searchHistoryVerification === "ledger") {
|
|
4978
|
+
if (!selected.searchHistory) assertCompleteSearchHistory(method.name, selected.searchHistory);
|
|
4979
|
+
verifySearchHistoryArtifact(selected.searchHistory, options.storage);
|
|
4980
|
+
}
|
|
4981
|
+
return {
|
|
4982
|
+
selected,
|
|
4983
|
+
cost: costScope.reconcile(selected.cost),
|
|
4984
|
+
history: options.searchHistoryVerification === "ledger" ? Object.freeze({
|
|
4985
|
+
...history,
|
|
4986
|
+
ledgerVerified: true
|
|
4987
|
+
}) : history
|
|
4988
|
+
};
|
|
4989
|
+
}
|
|
4990
|
+
function assertOptimizationResult(name, result) {
|
|
4991
|
+
if (!result || typeof result !== "object") throw new Error(`compareOptimizationMethods: method '${name}' returned no result`);
|
|
4992
|
+
try {
|
|
4993
|
+
surfaceContentHash(result.winnerSurface);
|
|
4994
|
+
} catch (cause) {
|
|
4995
|
+
throw new Error(`compareOptimizationMethods: method '${name}' returned an invalid winnerSurface`, { cause });
|
|
4996
|
+
}
|
|
4997
|
+
assertComparisonCost(result.cost, `method '${name}'`);
|
|
4998
|
+
if (result.durationMs !== void 0 && (!Number.isFinite(result.durationMs) || result.durationMs < 0)) throw new Error(`compareOptimizationMethods: method '${name}' returned an invalid durationMs`);
|
|
4999
|
+
if (result.provenance !== void 0) assertOptimizationProvenance(name, result.provenance);
|
|
5000
|
+
}
|
|
5001
|
+
function assertOptimizationProvenance(methodName, value) {
|
|
5002
|
+
const fail = (field) => {
|
|
5003
|
+
throw new Error(`compareOptimizationMethods: method '${methodName}' returned invalid provenance.${field}`);
|
|
5004
|
+
};
|
|
5005
|
+
if (!value || typeof value !== "object") fail("value");
|
|
5006
|
+
if (value.source?.kind !== "package" || !["observed", "declared"].includes(value.source.evidence) || typeof value.source.package !== "string" || !value.source.package.trim() || typeof value.source.version !== "string" || !value.source.version.trim()) fail("source");
|
|
5007
|
+
for (const [field, entry] of [["sourceUrl", value.source.sourceUrl], ["revision", value.source.revision]]) if (entry !== void 0 && (typeof entry !== "string" || !entry.trim())) fail(`source.${field}`);
|
|
5008
|
+
if (typeof value.runId !== "string" || !value.runId.trim()) fail("runId");
|
|
5009
|
+
if (value.optimizerModel !== void 0 && (typeof value.optimizerModel !== "string" || !value.optimizerModel.trim() || value.optimizerModel.trim() !== value.optimizerModel)) fail("optimizerModel");
|
|
5010
|
+
if (value.optimizerCallRef !== void 0 && (typeof value.optimizerCallRef !== "string" || !value.optimizerCallRef.trim() || value.optimizerCallRef.trim() !== value.optimizerCallRef)) fail("optimizerCallRef");
|
|
5011
|
+
if (typeof value.resumed !== "boolean") fail("resumed");
|
|
5012
|
+
if (value.seedApplied !== void 0 && typeof value.seedApplied !== "boolean") fail("seedApplied");
|
|
5013
|
+
if (!Number.isSafeInteger(value.evaluationCount) || value.evaluationCount < 0) fail("evaluationCount");
|
|
5014
|
+
if (typeof value.artifactDir !== "string" || !value.artifactDir.trim()) fail("artifactDir");
|
|
5015
|
+
if (value.tokenUsage !== void 0) {
|
|
5016
|
+
for (const field of [
|
|
5017
|
+
"inputTokens",
|
|
5018
|
+
"outputTokens",
|
|
5019
|
+
"totalTokens",
|
|
5020
|
+
"calls"
|
|
5021
|
+
]) if (!Number.isSafeInteger(value.tokenUsage[field]) || value.tokenUsage[field] < 0) fail(`tokenUsage.${field}`);
|
|
5022
|
+
for (const field of [
|
|
5023
|
+
"cachedInputTokens",
|
|
5024
|
+
"cacheWriteInputTokens",
|
|
5025
|
+
"reasoningTokens"
|
|
5026
|
+
]) {
|
|
5027
|
+
const entry = value.tokenUsage[field];
|
|
5028
|
+
if (entry !== void 0 && (!Number.isSafeInteger(entry) || entry < 0)) fail(`tokenUsage.${field}`);
|
|
5029
|
+
}
|
|
5030
|
+
if ((value.tokenUsage.cachedInputTokens ?? 0) + (value.tokenUsage.cacheWriteInputTokens ?? 0) > value.tokenUsage.inputTokens) fail("tokenUsage.inputTokens");
|
|
5031
|
+
if (value.tokenUsage.reasoningTokens !== void 0 && value.tokenUsage.reasoningTokens > value.tokenUsage.outputTokens) fail("tokenUsage.reasoningTokens");
|
|
5032
|
+
if (value.tokenUsage.totalTokens !== value.tokenUsage.inputTokens + value.tokenUsage.outputTokens) fail("tokenUsage.totalTokens");
|
|
5033
|
+
}
|
|
5034
|
+
if (value.observations !== void 0) {
|
|
5035
|
+
if (value.observations.scope !== "callback-submitted-candidates" || typeof value.observations.path !== "string" || !value.observations.path.trim() || typeof value.observations.sha256 !== "string" || !/^sha256:[0-9a-f]{64}$/.test(value.observations.sha256)) fail("observations");
|
|
5036
|
+
for (const field of [
|
|
5037
|
+
"submittedCandidates",
|
|
5038
|
+
"evaluations",
|
|
5039
|
+
"refusals"
|
|
5040
|
+
]) if (!Number.isSafeInteger(value.observations[field]) || value.observations[field] < 0) fail(`observations.${field}`);
|
|
5041
|
+
}
|
|
5042
|
+
if (value.gepaCandidatePopulation !== void 0) {
|
|
5043
|
+
try {
|
|
5044
|
+
assertGepaCandidatePopulationSummary(value.gepaCandidatePopulation);
|
|
5045
|
+
} catch {
|
|
5046
|
+
fail("gepaCandidatePopulation");
|
|
5047
|
+
}
|
|
5048
|
+
if (value.gepaCandidatePopulation.runId !== value.runId) fail("gepaCandidatePopulation.runId");
|
|
5049
|
+
}
|
|
5050
|
+
if (value.modelExecutions !== void 0) {
|
|
5051
|
+
if (value.modelExecutions.scope !== "runtime-model-calls" || typeof value.modelExecutions.path !== "string" || !value.modelExecutions.path.trim() || typeof value.modelExecutions.sha256 !== "string" || !/^sha256:[0-9a-f]{64}$/.test(value.modelExecutions.sha256)) fail("modelExecutions");
|
|
5052
|
+
for (const field of [
|
|
5053
|
+
"calls",
|
|
5054
|
+
"succeeded",
|
|
5055
|
+
"failed"
|
|
5056
|
+
]) if (!Number.isSafeInteger(value.modelExecutions[field]) || value.modelExecutions[field] < 0) fail(`modelExecutions.${field}`);
|
|
5057
|
+
if (value.modelExecutions.calls !== value.modelExecutions.succeeded + value.modelExecutions.failed) fail("modelExecutions.calls");
|
|
5058
|
+
}
|
|
5059
|
+
if (value.optimizerModel === void 0 !== (value.optimizerCallRef === void 0) || value.optimizerModel === void 0 !== (value.modelExecutions === void 0)) fail("optimizerModel execution provenance");
|
|
5060
|
+
}
|
|
5061
|
+
//#endregion
|
|
5527
5062
|
//#region src/campaign/presets/compare-optimization-methods.ts
|
|
5528
5063
|
/**
|
|
5529
5064
|
* Compare optimization methods on shared train, selection, and test data.
|
|
@@ -5538,7 +5073,7 @@ async function compareOptimizationMethods(opts) {
|
|
|
5538
5073
|
assertOptimizationMethods(opts.methods);
|
|
5539
5074
|
assertComparisonPartitions(opts);
|
|
5540
5075
|
const searchHistoryPolicy = opts.searchHistoryPolicy ?? "allow-missing";
|
|
5541
|
-
|
|
5076
|
+
assertSearchHistoryAdmissionOptions(opts);
|
|
5542
5077
|
const seed = opts.seed ?? 42;
|
|
5543
5078
|
const confidence = opts.confidence ?? .95;
|
|
5544
5079
|
assertConfidence(confidence);
|
|
@@ -5548,6 +5083,8 @@ async function compareOptimizationMethods(opts) {
|
|
|
5548
5083
|
const minimumResamples = minimumBootstrapResamples(confidence, comparisonCount);
|
|
5549
5084
|
const resamples = opts.resamples ?? Math.max(2e3, minimumResamples);
|
|
5550
5085
|
assertComparisonControls(opts, seed, resamples, confidence);
|
|
5086
|
+
const evidence = opts.evidence === void 0 ? void 0 : structuredClone(opts.evidence);
|
|
5087
|
+
const evidenceBySurface = /* @__PURE__ */ new Map();
|
|
5551
5088
|
const storage = opts.storage ?? fsCampaignStorage();
|
|
5552
5089
|
const resolvedRunDir = resolveRunDir(opts.runDir, opts.repo);
|
|
5553
5090
|
const baselineSurface = structuredClone(opts.baselineSurface);
|
|
@@ -5569,6 +5106,11 @@ async function compareOptimizationMethods(opts) {
|
|
|
5569
5106
|
runDir: `${resolvedRunDir}/${tag}`
|
|
5570
5107
|
});
|
|
5571
5108
|
assertCompleteCampaign(campaign, opts.testScenarios, opts.reps ?? 1, true, `compareOptimizationMethods: ${tag} final comparison`);
|
|
5109
|
+
if (evidence) evidenceBySurface.set(surfaceContentHash(measuredSurface), createCampaignEvidenceReceipt({
|
|
5110
|
+
campaign,
|
|
5111
|
+
surface: measuredSurface,
|
|
5112
|
+
context: evidence
|
|
5113
|
+
}));
|
|
5572
5114
|
const byScenario = {};
|
|
5573
5115
|
for (const { scenarioId, composite } of campaignBreakdown(campaign).scenarios) byScenario[scenarioId] = composite;
|
|
5574
5116
|
return byScenario;
|
|
@@ -5582,16 +5124,18 @@ async function compareOptimizationMethods(opts) {
|
|
|
5582
5124
|
const optimizationOwner = new AbortController();
|
|
5583
5125
|
const optimized = await mapConcurrent(opts.methods, optimizationConcurrency, async (method) => {
|
|
5584
5126
|
try {
|
|
5585
|
-
const
|
|
5586
|
-
|
|
5587
|
-
|
|
5588
|
-
|
|
5589
|
-
|
|
5590
|
-
|
|
5127
|
+
const { selected: out, cost, history: searchHistoryCoverage } = await executeOptimizationMethod({
|
|
5128
|
+
method,
|
|
5129
|
+
input: createOptimizationMethodInput(opts, method.name, resolvedRunDir, seed, baselineSurface, costLedger, optimizationOwner.signal),
|
|
5130
|
+
storage,
|
|
5131
|
+
searchHistoryPolicy: opts.searchHistoryPolicy,
|
|
5132
|
+
searchHistoryVerification: opts.searchHistoryVerification
|
|
5133
|
+
});
|
|
5134
|
+
const winnerSurface = out.winnerSurface;
|
|
5591
5135
|
return {
|
|
5592
5136
|
name: method.name,
|
|
5593
5137
|
winnerSurface,
|
|
5594
|
-
cost
|
|
5138
|
+
cost,
|
|
5595
5139
|
...out.durationMs === void 0 ? {} : { durationMs: out.durationMs },
|
|
5596
5140
|
...out.provenance === void 0 ? {} : { provenance: out.provenance },
|
|
5597
5141
|
searchHistoryCoverage
|
|
@@ -5647,6 +5191,15 @@ async function compareOptimizationMethods(opts) {
|
|
|
5647
5191
|
winnerSurface: structuredClone(w.winnerSurface),
|
|
5648
5192
|
rank: 0
|
|
5649
5193
|
};
|
|
5194
|
+
if (evidence) {
|
|
5195
|
+
const baseline = evidenceBySurface.get(surfaceContentHash(baselineSurface));
|
|
5196
|
+
const winner = evidenceBySurface.get(surfaceContentHash(w.winnerSurface));
|
|
5197
|
+
if (!baseline || !winner) throw new Error("final measurement evidence is missing");
|
|
5198
|
+
score.evidence = {
|
|
5199
|
+
baseline,
|
|
5200
|
+
winner
|
|
5201
|
+
};
|
|
5202
|
+
}
|
|
5650
5203
|
if (w.durationMs !== void 0) score.durationMs = w.durationMs;
|
|
5651
5204
|
if (w.provenance !== void 0) score.provenance = structuredClone(w.provenance);
|
|
5652
5205
|
return score;
|
|
@@ -5739,77 +5292,6 @@ function assertOptimizationMethods(methods) {
|
|
|
5739
5292
|
pathOwners.set(pathKey, method.name);
|
|
5740
5293
|
}
|
|
5741
5294
|
}
|
|
5742
|
-
function assertOptimizationResult(name, result) {
|
|
5743
|
-
if (!result || typeof result !== "object") throw new Error(`compareOptimizationMethods: method '${name}' returned no result`);
|
|
5744
|
-
try {
|
|
5745
|
-
surfaceContentHash(result.winnerSurface);
|
|
5746
|
-
} catch (cause) {
|
|
5747
|
-
throw new Error(`compareOptimizationMethods: method '${name}' returned an invalid winnerSurface`, { cause });
|
|
5748
|
-
}
|
|
5749
|
-
assertComparisonCost(result.cost, `method '${name}'`);
|
|
5750
|
-
if (result.durationMs !== void 0 && (!Number.isFinite(result.durationMs) || result.durationMs < 0)) throw new Error(`compareOptimizationMethods: method '${name}' returned an invalid durationMs`);
|
|
5751
|
-
if (result.provenance !== void 0) assertOptimizationProvenance(name, result.provenance);
|
|
5752
|
-
}
|
|
5753
|
-
function assertOptimizationProvenance(methodName, value) {
|
|
5754
|
-
const fail = (field) => {
|
|
5755
|
-
throw new Error(`compareOptimizationMethods: method '${methodName}' returned invalid provenance.${field}`);
|
|
5756
|
-
};
|
|
5757
|
-
if (!value || typeof value !== "object") fail("value");
|
|
5758
|
-
if (value.source?.kind !== "package" || !["observed", "declared"].includes(value.source.evidence) || typeof value.source.package !== "string" || !value.source.package.trim() || typeof value.source.version !== "string" || !value.source.version.trim()) fail("source");
|
|
5759
|
-
for (const [field, entry] of [["sourceUrl", value.source.sourceUrl], ["revision", value.source.revision]]) if (entry !== void 0 && (typeof entry !== "string" || !entry.trim())) fail(`source.${field}`);
|
|
5760
|
-
if (typeof value.runId !== "string" || !value.runId.trim()) fail("runId");
|
|
5761
|
-
if (value.optimizerModel !== void 0 && (typeof value.optimizerModel !== "string" || !value.optimizerModel.trim() || value.optimizerModel.trim() !== value.optimizerModel)) fail("optimizerModel");
|
|
5762
|
-
if (value.optimizerCallRef !== void 0 && (typeof value.optimizerCallRef !== "string" || !value.optimizerCallRef.trim() || value.optimizerCallRef.trim() !== value.optimizerCallRef)) fail("optimizerCallRef");
|
|
5763
|
-
if (typeof value.resumed !== "boolean") fail("resumed");
|
|
5764
|
-
if (value.seedApplied !== void 0 && typeof value.seedApplied !== "boolean") fail("seedApplied");
|
|
5765
|
-
if (!Number.isSafeInteger(value.evaluationCount) || value.evaluationCount < 0) fail("evaluationCount");
|
|
5766
|
-
if (typeof value.artifactDir !== "string" || !value.artifactDir.trim()) fail("artifactDir");
|
|
5767
|
-
if (value.tokenUsage !== void 0) {
|
|
5768
|
-
for (const field of [
|
|
5769
|
-
"inputTokens",
|
|
5770
|
-
"outputTokens",
|
|
5771
|
-
"totalTokens",
|
|
5772
|
-
"calls"
|
|
5773
|
-
]) if (!Number.isSafeInteger(value.tokenUsage[field]) || value.tokenUsage[field] < 0) fail(`tokenUsage.${field}`);
|
|
5774
|
-
for (const field of [
|
|
5775
|
-
"cachedInputTokens",
|
|
5776
|
-
"cacheWriteInputTokens",
|
|
5777
|
-
"reasoningTokens"
|
|
5778
|
-
]) {
|
|
5779
|
-
const entry = value.tokenUsage[field];
|
|
5780
|
-
if (entry !== void 0 && (!Number.isSafeInteger(entry) || entry < 0)) fail(`tokenUsage.${field}`);
|
|
5781
|
-
}
|
|
5782
|
-
if ((value.tokenUsage.cachedInputTokens ?? 0) + (value.tokenUsage.cacheWriteInputTokens ?? 0) > value.tokenUsage.inputTokens) fail("tokenUsage.inputTokens");
|
|
5783
|
-
if (value.tokenUsage.reasoningTokens !== void 0 && value.tokenUsage.reasoningTokens > value.tokenUsage.outputTokens) fail("tokenUsage.reasoningTokens");
|
|
5784
|
-
if (value.tokenUsage.totalTokens !== value.tokenUsage.inputTokens + value.tokenUsage.outputTokens) fail("tokenUsage.totalTokens");
|
|
5785
|
-
}
|
|
5786
|
-
if (value.observations !== void 0) {
|
|
5787
|
-
if (value.observations.scope !== "callback-submitted-candidates" || typeof value.observations.path !== "string" || !value.observations.path.trim() || typeof value.observations.sha256 !== "string" || !/^sha256:[0-9a-f]{64}$/.test(value.observations.sha256)) fail("observations");
|
|
5788
|
-
for (const field of [
|
|
5789
|
-
"submittedCandidates",
|
|
5790
|
-
"evaluations",
|
|
5791
|
-
"refusals"
|
|
5792
|
-
]) if (!Number.isSafeInteger(value.observations[field]) || value.observations[field] < 0) fail(`observations.${field}`);
|
|
5793
|
-
}
|
|
5794
|
-
if (value.gepaCandidatePopulation !== void 0) {
|
|
5795
|
-
try {
|
|
5796
|
-
assertGepaCandidatePopulationSummary(value.gepaCandidatePopulation);
|
|
5797
|
-
} catch {
|
|
5798
|
-
fail("gepaCandidatePopulation");
|
|
5799
|
-
}
|
|
5800
|
-
if (value.gepaCandidatePopulation.runId !== value.runId) fail("gepaCandidatePopulation.runId");
|
|
5801
|
-
}
|
|
5802
|
-
if (value.modelExecutions !== void 0) {
|
|
5803
|
-
if (value.modelExecutions.scope !== "runtime-model-calls" || typeof value.modelExecutions.path !== "string" || !value.modelExecutions.path.trim() || typeof value.modelExecutions.sha256 !== "string" || !/^sha256:[0-9a-f]{64}$/.test(value.modelExecutions.sha256)) fail("modelExecutions");
|
|
5804
|
-
for (const field of [
|
|
5805
|
-
"calls",
|
|
5806
|
-
"succeeded",
|
|
5807
|
-
"failed"
|
|
5808
|
-
]) if (!Number.isSafeInteger(value.modelExecutions[field]) || value.modelExecutions[field] < 0) fail(`modelExecutions.${field}`);
|
|
5809
|
-
if (value.modelExecutions.calls !== value.modelExecutions.succeeded + value.modelExecutions.failed) fail("modelExecutions.calls");
|
|
5810
|
-
}
|
|
5811
|
-
if (value.optimizerModel === void 0 !== (value.optimizerCallRef === void 0) || value.optimizerModel === void 0 !== (value.modelExecutions === void 0)) fail("optimizerModel execution provenance");
|
|
5812
|
-
}
|
|
5813
5295
|
function assertComparisonControls(opts, seed, resamples, confidence) {
|
|
5814
5296
|
if (opts.optimizationRunOptions && "costCeiling" in opts.optimizationRunOptions) throw new Error("compareOptimizationMethods: optimizationRunOptions.costCeiling is not supported; costCeiling covers optimization and final scoring");
|
|
5815
5297
|
if (!opts.judges || opts.judges.length === 0) throw new Error("compareOptimizationMethods: at least one judge is required");
|
|
@@ -5899,18 +5381,13 @@ function minimumBootstrapResamples(confidence, comparisonCount) {
|
|
|
5899
5381
|
}
|
|
5900
5382
|
function createOptimizationMethodInput(opts, methodName, resolvedRunDir, seed, baselineSurface, costLedger, optimizationSignal) {
|
|
5901
5383
|
const methodRunDir = `${resolvedRunDir}/optimization/${slug(methodName)}`;
|
|
5902
|
-
const cloneScenarios = (scenarios) => Object.freeze(scenarios.map((scenario) => structuredClone(scenario)));
|
|
5903
|
-
const judges = opts.judges.map((judge) => Object.freeze({
|
|
5904
|
-
...judge,
|
|
5905
|
-
dimensions: Object.freeze(judge.dimensions.map((dimension) => Object.freeze({ ...dimension })))
|
|
5906
|
-
}));
|
|
5907
5384
|
const signal = combineAbortSignals(opts.signal, opts.optimizationRunOptions?.signal, optimizationSignal);
|
|
5908
5385
|
return Object.freeze({
|
|
5909
5386
|
baselineSurface: structuredClone(baselineSurface),
|
|
5910
|
-
trainScenarios:
|
|
5911
|
-
selectionScenarios:
|
|
5387
|
+
trainScenarios: opts.trainScenarios,
|
|
5388
|
+
selectionScenarios: opts.selectionScenarios,
|
|
5912
5389
|
dispatchWithSurface: opts.dispatchWithSurface,
|
|
5913
|
-
judges:
|
|
5390
|
+
judges: opts.judges,
|
|
5914
5391
|
runDir: methodRunDir,
|
|
5915
5392
|
seed,
|
|
5916
5393
|
runOptions: Object.freeze({
|
|
@@ -6440,6 +5917,6 @@ function firstString(value) {
|
|
|
6440
5917
|
return typeof value === "string" && value.trim() ? value : void 0;
|
|
6441
5918
|
}
|
|
6442
5919
|
//#endregion
|
|
6443
|
-
export {
|
|
5920
|
+
export { openSearchLedger as A, redTeamDataset as B, assertSearchHistoryAdmissionOptions as C, verifySearchHistoryArtifact as D, searchHistoryCoverageRow as E, paretoFrontierWithCrowding as F, runCampaign as G, scoreRedTeamOutput as H, runFinalComparison as I, tangleTracesRoot as J, planCampaignRun as K, openAutoPr as L, validateSearchLedgerEvent as M, dominates as N, verifySearchHistoryReceipt as O, paretoFrontier as P, computeManifestHash as Q, defaultProductionGate as R, assertCompleteSearchHistory as S, createSearchHistoryReceipt as T, runCanaries as U, redTeamReport as V, runEval as W, buildCellSchedule as X, readCachedCell as Y, cellCachePath as Z, isProposedCandidate as _, transientDispatchFailure as a, recordCandidatePopulationSearch as b, compareOptimizationMethods as c, combineComparisonCosts as d, costFromLedgerSummary as f, runOptimization as g, runImprovementLoop as h, quotaExhaustedUntil as i, replaySearchLedgerText as j, FileSearchLedger as k, optimizationTokenUsageFromSummary as l, readGepaCandidatePopulationArtifact as m, JudgeParseError as n, aggregateRunScore as o, assertGepaCandidatePopulationSummary as p, resolveRunDir as q, isTransientTransportFailure as r, clamp01 as s, llmJudge as t, executeOptimizationMethod as u, labelTrustRank as v, assertSearchHistoryMatchesReplay as w, SearchHistoryRequiredError as x, SearchRecorder as y, DEFAULT_RED_TEAM_CORPUS as z };
|
|
6444
5921
|
|
|
6445
|
-
//# sourceMappingURL=llm-judge-
|
|
5922
|
+
//# sourceMappingURL=llm-judge-DliimmRb.js.map
|