@tangle-network/agent-eval 0.174.0 → 0.176.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/CHANGELOG.md +37 -0
  2. package/README.md +1 -1
  3. package/dist/adapters/http.d.ts +1 -1
  4. package/dist/agent-profile-cell-0gSi5ffD.js +374 -0
  5. package/dist/agent-profile-cell-0gSi5ffD.js.map +1 -0
  6. package/dist/analyst/index.d.ts +3 -3
  7. package/dist/analyst/index.js +4 -4
  8. package/dist/{benchmark-command-mZIlR-ra.js → benchmark-command-yPqjcZnC.js} +7 -7
  9. package/dist/{benchmark-command-mZIlR-ra.js.map → benchmark-command-yPqjcZnC.js.map} +1 -1
  10. package/dist/benchmarks/index.d.ts +1 -1
  11. package/dist/benchmarks/index.js +3 -3
  12. package/dist/campaign/index.d.ts +3 -3
  13. package/dist/campaign/index.js +9 -9
  14. package/dist/{campaign-BzMSCejE.js → campaign-85igdlgG.js} +12 -12
  15. package/dist/{campaign-BzMSCejE.js.map → campaign-85igdlgG.js.map} +1 -1
  16. package/dist/campaign-evidence-D8DBLqLI.js +2083 -0
  17. package/dist/campaign-evidence-D8DBLqLI.js.map +1 -0
  18. package/dist/{opencode-sqlite-eK6HW6dr.js → claude-jsonl-CxZZrDJ3.js} +9 -149
  19. package/dist/claude-jsonl-CxZZrDJ3.js.map +1 -0
  20. package/dist/cli.js +9 -2
  21. package/dist/cli.js.map +1 -1
  22. package/dist/contract/index.d.ts +4 -4
  23. package/dist/contract/index.js +9 -9
  24. package/dist/{default-registry-CrAp0pYq.js → default-registry-DBqVI4pq.js} +2 -2
  25. package/dist/{default-registry-CrAp0pYq.js.map → default-registry-DBqVI4pq.js.map} +1 -1
  26. package/dist/{define-agent-eval-V1jQyCDR.d.ts → define-agent-eval-CCbl8k2E.d.ts} +11 -4
  27. package/dist/define-agent-eval-CCbl8k2E.d.ts.map +1 -0
  28. package/dist/{define-agent-eval-ox5McL6e.js → define-agent-eval-DEMsu5eA.js} +54 -36
  29. package/dist/define-agent-eval-DEMsu5eA.js.map +1 -0
  30. package/dist/{dspy-rlm-engine-Caz2pl4L.js → dspy-rlm-engine-DqjER2sV.js} +2 -2
  31. package/dist/{dspy-rlm-engine-Caz2pl4L.js.map → dspy-rlm-engine-DqjER2sV.js.map} +1 -1
  32. package/dist/{eval-campaign-BeAjdhzC.js → eval-campaign-Cs-7MiCs.js} +4 -5
  33. package/dist/{eval-campaign-BeAjdhzC.js.map → eval-campaign-Cs-7MiCs.js.map} +1 -1
  34. package/dist/experiment/index.d.ts +3 -68
  35. package/dist/experiment/index.d.ts.map +1 -1
  36. package/dist/experiment/index.js +6 -128
  37. package/dist/experiment/index.js.map +1 -1
  38. package/dist/{attestation-XSUpbc4o.js → experiment-tracker-BKEumQug.js} +2 -96
  39. package/dist/experiment-tracker-BKEumQug.js.map +1 -0
  40. package/dist/{attestation-c1QvaBdX.d.ts → experiment-tracker-CNwqCZFD.d.ts} +2 -78
  41. package/dist/experiment-tracker-CNwqCZFD.d.ts.map +1 -0
  42. package/dist/{external-optimizer-process-CxnFL1hd.js → external-optimizer-process-Dlz8YxrT.js} +3 -3
  43. package/dist/{external-optimizer-process-CxnFL1hd.js.map → external-optimizer-process-Dlz8YxrT.js.map} +1 -1
  44. package/dist/{external-optimizer-subprocess-CQi27uEI.js → external-optimizer-subprocess-q3VzlGAO.js} +2 -2
  45. package/dist/{external-optimizer-subprocess-CQi27uEI.js.map → external-optimizer-subprocess-q3VzlGAO.js.map} +1 -1
  46. package/dist/{index-Bn-nlnSV.d.ts → index-BAAiSF3_.d.ts} +2 -2
  47. package/dist/{index-Bn-nlnSV.d.ts.map → index-BAAiSF3_.d.ts.map} +1 -1
  48. package/dist/{index-DKXuBPXf.d.ts → index-Bg6OT2Dd.d.ts} +23 -10
  49. package/dist/{index-DKXuBPXf.d.ts.map → index-Bg6OT2Dd.d.ts.map} +1 -1
  50. package/dist/{index-BTrx5s8m.d.ts → index-DBkcm_9H.d.ts} +4 -4
  51. package/dist/{index-BTrx5s8m.d.ts.map → index-DBkcm_9H.d.ts.map} +1 -1
  52. package/dist/{index-D-UdhAmg.d.ts → index-u0d1Jp4F.d.ts} +4 -2
  53. package/dist/{index-D-UdhAmg.d.ts.map → index-u0d1Jp4F.d.ts.map} +1 -1
  54. package/dist/index.d.ts +5 -5
  55. package/dist/index.js +15 -16
  56. package/dist/index.js.map +1 -1
  57. package/dist/{integrity-BWywb34E.js → integrity-DsHWCebQ.js} +11 -435
  58. package/dist/integrity-DsHWCebQ.js.map +1 -0
  59. package/dist/ledger-core/index.d.ts +2 -2
  60. package/dist/ledger-core/index.js +2 -2
  61. package/dist/{ledger-core-PIfjCbKn.js → ledger-core-Cs9f7385.js} +60 -47
  62. package/dist/{ledger-core-PIfjCbKn.js.map → ledger-core-Cs9f7385.js.map} +1 -1
  63. package/dist/{llm-judge-DmNaBrXB.js → llm-judge-DliimmRb.js} +994 -1517
  64. package/dist/llm-judge-DliimmRb.js.map +1 -0
  65. package/dist/{mint-vWOdD8Ae.js → mint-Cc1_zwRQ.js} +2 -2
  66. package/dist/{mint-vWOdD8Ae.js.map → mint-Cc1_zwRQ.js.map} +1 -1
  67. package/dist/openapi.json +1 -1
  68. package/dist/opencode-sqlite-CNw3vubS.js +145 -0
  69. package/dist/opencode-sqlite-CNw3vubS.js.map +1 -0
  70. package/dist/{produced-state-B8mw6zj9.js → produced-state-DrMqa2HD.js} +3 -2
  71. package/dist/{produced-state-B8mw6zj9.js.map → produced-state-DrMqa2HD.js.map} +1 -1
  72. package/dist/profile-cell.js +1 -268
  73. package/dist/{promotion-policy-LY9mVQ7W.js → promotion-policy-DWOm70gx.js} +2 -2
  74. package/dist/{promotion-policy-LY9mVQ7W.js.map → promotion-policy-DWOm70gx.js.map} +1 -1
  75. package/dist/{release-confidence-BsGEg_xg.js → release-confidence-BcGCclTB.js} +2 -2
  76. package/dist/{release-confidence-BsGEg_xg.js.map → release-confidence-BcGCclTB.js.map} +1 -1
  77. package/dist/report-command-DKlXfU5r.js +1528 -0
  78. package/dist/report-command-DKlXfU5r.js.map +1 -0
  79. package/dist/reporting.js +2 -2
  80. package/dist/{reward-hacking-CKW4teig.js → reward-hacking-D0XwhVWE.js} +2 -215
  81. package/dist/reward-hacking-D0XwhVWE.js.map +1 -0
  82. package/dist/rl.js +5 -4
  83. package/dist/rl.js.map +1 -1
  84. package/dist/rollout/index.js +4 -3
  85. package/dist/{rollout-C-znbbYg.js → rollout-DmoJVqrF.js} +4 -3
  86. package/dist/{rollout-C-znbbYg.js.map → rollout-DmoJVqrF.js.map} +1 -1
  87. package/dist/run-record-CR63CpHK.js +216 -0
  88. package/dist/run-record-CR63CpHK.js.map +1 -0
  89. package/dist/{run-record-ZIsR9Fif.js → run-record-DQpSf7t-.js} +2 -2
  90. package/dist/{run-record-ZIsR9Fif.js.map → run-record-DQpSf7t-.js.map} +1 -1
  91. package/dist/{semantic-concept-judge-E3s_fEjB.js → semantic-concept-judge-Dw-f7TEs.js} +3 -3
  92. package/dist/{semantic-concept-judge-E3s_fEjB.js.map → semantic-concept-judge-Dw-f7TEs.js.map} +1 -1
  93. package/dist/{sequential-B51qAYE4.js → sequential-B5gXgcyp.js} +3 -3
  94. package/dist/{sequential-B51qAYE4.js.map → sequential-B5gXgcyp.js.map} +1 -1
  95. package/dist/{skillopt-optimization-method-f7399oGb.js → skillopt-optimization-method-CV7go7ex.js} +6 -749
  96. package/dist/skillopt-optimization-method-CV7go7ex.js.map +1 -0
  97. package/dist/{statistical-heldout-Cqb73yE9.d.ts → statistical-heldout-Z9NROFFS.d.ts} +156 -3
  98. package/dist/statistical-heldout-Z9NROFFS.d.ts.map +1 -0
  99. package/dist/{summary-report-Bgh8CpNK.js → summary-report-B16xy9Kd.js} +2 -2
  100. package/dist/{summary-report-Bgh8CpNK.js.map → summary-report-B16xy9Kd.js.map} +1 -1
  101. package/dist/supervisor-run/index.d.ts +71 -6
  102. package/dist/supervisor-run/index.d.ts.map +1 -1
  103. package/dist/supervisor-run/index.js +6 -1357
  104. package/dist/supervisor-run/index.js.map +1 -1
  105. package/dist/terminal-record-Ce9_UjRz.js +539 -0
  106. package/dist/terminal-record-Ce9_UjRz.js.map +1 -0
  107. package/dist/traces.js +1 -1
  108. package/dist/{types-CoPUTiXb.d.ts → types-vUdAx2Cj.d.ts} +65 -3
  109. package/dist/types-vUdAx2Cj.d.ts.map +1 -0
  110. package/docs/public-api.md +62 -39
  111. package/docs/search-history-receipts.md +48 -1
  112. package/package.json +1 -1
  113. package/dist/attestation-XSUpbc4o.js.map +0 -1
  114. package/dist/attestation-c1QvaBdX.d.ts.map +0 -1
  115. package/dist/define-agent-eval-V1jQyCDR.d.ts.map +0 -1
  116. package/dist/define-agent-eval-ox5McL6e.js.map +0 -1
  117. package/dist/integrity-BWywb34E.js.map +0 -1
  118. package/dist/llm-judge-DmNaBrXB.js.map +0 -1
  119. package/dist/opencode-sqlite-eK6HW6dr.js.map +0 -1
  120. package/dist/power-preflight-CFXm0Vjo.js +0 -502
  121. package/dist/power-preflight-CFXm0Vjo.js.map +0 -1
  122. package/dist/pre-registration-D94b7Of5.js +0 -110
  123. package/dist/pre-registration-D94b7Of5.js.map +0 -1
  124. package/dist/profile-cell.js.map +0 -1
  125. package/dist/reward-hacking-CKW4teig.js.map +0 -1
  126. package/dist/skillopt-optimization-method-f7399oGb.js.map +0 -1
  127. package/dist/statistical-heldout-Cqb73yE9.d.ts.map +0 -1
  128. package/dist/types-CoPUTiXb.d.ts.map +0 -1
@@ -1,15 +1,16 @@
1
- import { i as JudgeError, s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
1
+ import { i as JudgeError, s as ValidationError } from "./errors-Dngq5h35.js";
2
2
  import { a as hashCanonical, i as compareCodeUnits, r as canonicalString } from "./canonical-DPyQ_rpt.js";
3
3
  import { o as summarizeNumberSeries, s as weightedComposite, t as confidenceInterval } from "./descriptive-1V17A-qa.js";
4
4
  import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
5
- import { i as modelHasSnapshot } from "./run-record-ZIsR9Fif.js";
5
+ import { i as modelHasSnapshot } from "./run-record-DQpSf7t-.js";
6
6
  import { n as contentHash } from "./verdict-cache-B3eCVQtY.js";
7
- import { a as campaignCellCostProvenance, o as campaignCellExecutionEvidence, t as detectRewardHacking, u as projectCampaignCellQuality } from "./reward-hacking-CKW4teig.js";
7
+ import { o as projectCampaignCellQuality, t as campaignCellCostProvenance } from "./run-record-CR63CpHK.js";
8
8
  import { i as CostLedger, t as CostAccountingIncompleteError } from "./cost-ledger-B1qx30B4.js";
9
- import { u as mapConcurrent } from "./ledger-core-PIfjCbKn.js";
10
- import { S as fsCampaignStorage, g as isRecord, h as isExternalTextCandidate, x as createRunCostLedger } from "./external-optimizer-subprocess-CQi27uEI.js";
9
+ import { $ as campaignScenarioIdentity, C as campaignMeanCompositeOrNull, G as surfaceHashMatches, H as surfaceContentHash, K as BackendIntegrityError, M as heldoutSignificance, N as pairHoldout, Q as campaignCoverage, S as campaignMeanComposite, U as surfaceDispatchRef, V as renderSurfaceDiff, W as surfaceHash, X as assertCampaignSplitIdentity, Y as assertCampaignDesign, Z as assertCompleteCampaign, b as assertFiniteRankKey, et as campaignSplitDigest, j as dimensionRegressions, nt as formatCoverageFailures, t as createCampaignEvidenceReceipt, w as compareRankKeys, x as campaignBreakdown } from "./campaign-evidence-D8DBLqLI.js";
10
+ import { d as mapConcurrent, n as replayLedgerText, t as FileLedgerJournal } from "./ledger-core-Cs9f7385.js";
11
+ import { D as SearchLedgerIntegrityError, E as SearchLedgerError, S as fsCampaignStorage, T as SearchLedgerConflictError, g as isRecord, h as isExternalTextCandidate, w as SEARCH_LEDGER_FILE_CONTEXT, x as createRunCostLedger } from "./external-optimizer-subprocess-q3VzlGAO.js";
11
12
  import { t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
12
- import { a as heldoutSignificance, i as dimensionRegressions, o as pairHoldout } from "./power-preflight-CFXm0Vjo.js";
13
+ import { t as detectRewardHacking } from "./reward-hacking-D0XwhVWE.js";
13
14
  import { n as assertProposalFindings, t as combineAbortSignals } from "./abort-signal-CtzAM_sJ.js";
14
15
  import { l as deepFreezeCanonicalJson } from "./types-DQ0e2E7y.js";
15
16
  import { d as stripFencedJson, o as costReceiptFromLlm, s as costReceiptFromLlmError, u as maximumChargeForLlmRequest } from "./llm-client-CGlSi8sb.js";
@@ -19,6 +20,7 @@ import { z } from "zod";
19
20
  import { basename, isAbsolute, join } from "node:path";
20
21
  import { homedir, tmpdir } from "node:os";
21
22
  import { execSync } from "node:child_process";
23
+ import { fileURLToPath } from "node:url";
22
24
  //#region src/campaign/campaign-manifest.ts
23
25
  /**
24
26
  * Campaign identity: the manifest hash over (scenarios, judges, dispatch,
@@ -286,251 +288,6 @@ function invalidCachedCellsError(cells) {
286
288
  return new CostAccountingIncompleteError(`runCampaign: cached cell(s) require explicit paid re-dispatch: ${cells.map((cell) => `${cell.cellId} (${cell.reason}${cell.detail ? `: ${cell.detail}` : ""})`).join(", ")}; refusing to begin campaign. Inspect planCampaignRun, then set rerunInvalidCachedCells: true to rerun only these cells while retaining valid caches, or resumable: false when a full rerun is intended.`);
287
289
  }
288
290
  //#endregion
289
- //#region src/campaign/coverage.ts
290
- /** Reject campaign designs whose denominator cannot be identified exactly. */
291
- function assertCampaignDesign(scenarios, reps) {
292
- if (!Number.isSafeInteger(reps) || reps < 1) throw new Error("campaign design requires reps to be a positive safe integer");
293
- const scenarioIds = /* @__PURE__ */ new Set();
294
- for (const scenario of scenarios) {
295
- if (typeof scenario.id !== "string" || scenario.id.trim().length === 0) throw new Error("campaign design requires every scenario to have a non-empty id");
296
- if (scenarioIds.has(scenario.id)) throw new Error(`campaign design contains duplicate scenario id '${scenario.id}'`);
297
- if (typeof scenario.kind !== "string" || scenario.kind.trim().length === 0) throw new Error("campaign design requires every scenario to have a non-empty kind");
298
- if (scenario.seedGroup !== void 0 && (typeof scenario.seedGroup !== "string" || scenario.seedGroup.trim().length === 0)) throw new Error("campaign design requires seedGroup to be a non-empty string when set");
299
- scenarioIds.add(scenario.id);
300
- }
301
- }
302
- /** Redacted but independently verifiable identity of one complete scenario. */
303
- function campaignScenarioIdentity(scenario) {
304
- assertCampaignDesign([scenario], 1);
305
- return {
306
- id: scenario.id,
307
- kind: scenario.kind,
308
- scenarioDigest: `sha256:${contentHash(scenario)}`
309
- };
310
- }
311
- /** Canonical split identity reconstructed from redacted scenario identities. */
312
- function campaignSplitDigestFromIdentities(scenarios, reps) {
313
- assertCampaignDesign(scenarios, reps);
314
- for (const scenario of scenarios) if (!/^sha256:[a-f0-9]{64}$/.test(scenario.scenarioDigest)) throw new Error(`campaign scenario '${scenario.id}' has an invalid digest`);
315
- return `sha256:${contentHash({
316
- schema: "tangle.campaign-split",
317
- scenarios: scenarios.map(({ id, kind, scenarioDigest }) => ({
318
- id,
319
- kind,
320
- scenarioDigest
321
- })),
322
- reps
323
- })}`;
324
- }
325
- /** Canonical identity of the exact scenario payloads and replicate count. */
326
- function campaignSplitDigest(scenarios, reps) {
327
- assertCampaignDesign(scenarios, reps);
328
- return campaignSplitDigestFromIdentities(scenarios.map(campaignScenarioIdentity), reps);
329
- }
330
- /** Refuse a campaign whose retained task identities contradict its split digest. */
331
- function assertCampaignSplitIdentity(scenarios, reps, splitDigest) {
332
- if (campaignSplitDigestFromIdentities(scenarios, reps) !== splitDigest) throw new Error("campaign split digest does not match its retained scenario identities");
333
- }
334
- /** Exact designed-denominator receipt for one campaign. */
335
- function campaignCoverage(cells, scenarios, reps, requireJudgeScore) {
336
- assertCampaignDesign(scenarios, reps);
337
- const expectedCellIds = designedCellIds(scenarios, reps);
338
- const cellsById = /* @__PURE__ */ new Map();
339
- for (const cell of cells) {
340
- const matches = cellsById.get(cell.cellId) ?? [];
341
- matches.push(cell);
342
- cellsById.set(cell.cellId, matches);
343
- }
344
- const scorableCellIds = [];
345
- const unscorableCells = [];
346
- for (const cellId of expectedCellIds) {
347
- const matches = cellsById.get(cellId) ?? [];
348
- if (matches.length === 0) {
349
- unscorableCells.push({
350
- cellId,
351
- reason: "missing campaign cell"
352
- });
353
- continue;
354
- }
355
- if (matches.length > 1) {
356
- unscorableCells.push({
357
- cellId,
358
- reason: `duplicate campaign cell (${matches.length})`
359
- });
360
- continue;
361
- }
362
- const cell = matches[0];
363
- const scoreEntries = Object.entries(cell.judgeScores);
364
- const successfulScores = scoreEntries.map(([, score]) => score).filter((score) => score.failed !== true && Number.isFinite(score.composite));
365
- const nonFiniteScores = scoreEntries.filter(([, score]) => score.failed !== true && (!Number.isFinite(score.composite) || Object.values(score.dimensions).some((value) => !Number.isFinite(value))));
366
- const reasons = [];
367
- if (cell.error) reasons.push(cell.error);
368
- if (cell.artifact === null || cell.artifact === void 0) reasons.push("missing artifact");
369
- if (!cell.error && requireJudgeScore && successfulScores.length === 0) reasons.push("no successful finite judge score");
370
- if (scoreEntries.some(([, score]) => score.failed === true)) reasons.push("judge score marked failed");
371
- const failedPanelJudges = [...new Set(scoreEntries.flatMap(([, score]) => score.failedJudges ?? []))].sort();
372
- if (failedPanelJudges.length > 0) reasons.push(`judge panel incomplete: ${failedPanelJudges.join(", ")}`);
373
- if (nonFiniteScores.length > 0) reasons.push(`non-finite judge score: ${nonFiniteScores.map(([name]) => name).sort().join(", ")}`);
374
- if (reasons.length > 0) unscorableCells.push({
375
- cellId,
376
- reason: reasons.join("; ")
377
- });
378
- else scorableCellIds.push(cellId);
379
- }
380
- const expected = new Set(expectedCellIds);
381
- for (const cell of cells) {
382
- if (cell.cellId !== `${cell.scenarioId}:${cell.rep}`) {
383
- unscorableCells.push({
384
- cellId: cell.cellId,
385
- reason: "campaign cell id does not match scenario id and rep"
386
- });
387
- continue;
388
- }
389
- if (!expected.has(cell.cellId)) unscorableCells.push({
390
- cellId: cell.cellId,
391
- reason: "unexpected campaign cell"
392
- });
393
- }
394
- return {
395
- complete: unscorableCells.length === 0 && scorableCellIds.length === expectedCellIds.length,
396
- expectedCellIds,
397
- scorableCellIds,
398
- unscorableCells
399
- };
400
- }
401
- function formatCoverageFailures(coverage) {
402
- const shown = coverage.unscorableCells.slice(0, 3).map((cell) => `${cell.cellId}: ${cell.reason}`).join("; ");
403
- const remainder = coverage.unscorableCells.length - Math.min(3, coverage.unscorableCells.length);
404
- return remainder > 0 ? `${shown}; +${remainder} more` : shown || "unknown coverage failure";
405
- }
406
- function designedCellIds(scenarios, reps) {
407
- const ids = [];
408
- for (const scenario of scenarios) for (let rep = 0; rep < reps; rep++) ids.push(`${scenario.id}:${rep}`);
409
- return ids;
410
- }
411
- /** Require the complete designed denominator before a final comparison. */
412
- function assertCompleteCampaign(campaign, scenarios, reps, requireJudgeScore, label) {
413
- const coverage = campaignCoverage(campaign.cells, scenarios, reps, requireJudgeScore);
414
- if (!coverage.complete) throw new Error(`${label} is incomplete (${coverage.scorableCellIds.length}/${coverage.expectedCellIds.length} designed cells scorable) — ${formatCoverageFailures(coverage)}. Refusing to compare unequal results.`);
415
- }
416
- //#endregion
417
- //#region src/integrity/backend-integrity.ts
418
- /**
419
- * Error thrown when an integrity assertion fails. Caller can pattern-match
420
- * by `code === 'AGENT_EVAL_BACKEND_STUB'` to differentiate from other
421
- * errors.
422
- */
423
- var BackendIntegrityError = class extends AgentEvalError {
424
- report;
425
- constructor(message, report) {
426
- super("backend_integrity", message);
427
- this.report = report;
428
- }
429
- };
430
- /**
431
- * Inspect a batch of RunRecords and return an integrity report. Pure
432
- * function — no I/O, no logging. The caller decides what to do with the
433
- * verdict (print warning, throw, gate CI, etc.).
434
- */
435
- function summarizeBackendIntegrity(records) {
436
- return summarizeBackendUsage(records.map((record) => ({
437
- inputTokens: record.tokenUsage.input,
438
- outputTokens: record.tokenUsage.output,
439
- costUsd: record.costUsd
440
- })));
441
- }
442
- /** Inspect settled agent calls from the canonical cost ledger. */
443
- function summarizeAgentReceiptIntegrity(receipts) {
444
- return summarizeBackendUsage(receipts.filter((receipt) => receipt.channel === "agent").map((receipt) => ({
445
- inputTokens: receipt.inputTokens,
446
- outputTokens: receipt.outputTokens,
447
- costUsd: receipt.costUsd
448
- })));
449
- }
450
- function summarizeBackendUsage(records) {
451
- const totalRecords = records.length;
452
- let stubRecords = 0;
453
- let realRecords = 0;
454
- let uncostedRecords = 0;
455
- let totalInputTokens = 0;
456
- let totalOutputTokens = 0;
457
- let totalCostUsd = 0;
458
- for (const rec of records) {
459
- totalInputTokens += rec.inputTokens;
460
- totalOutputTokens += rec.outputTokens;
461
- totalCostUsd += rec.costUsd ?? 0;
462
- if (rec.inputTokens === 0 && rec.outputTokens === 0) stubRecords++;
463
- else realRecords++;
464
- if (rec.outputTokens > 0 && (rec.costUsd === null || rec.costUsd === 0)) uncostedRecords++;
465
- }
466
- const verdict = totalRecords === 0 ? "stub" : stubRecords === totalRecords ? "stub" : stubRecords === 0 ? "real" : "mixed";
467
- const diagnosis = buildDiagnosis({
468
- totalRecords,
469
- stubRecords,
470
- realRecords,
471
- uncostedRecords,
472
- totalInputTokens,
473
- totalOutputTokens,
474
- totalCostUsd,
475
- verdict
476
- });
477
- return {
478
- totalRecords,
479
- stubRecords,
480
- realRecords,
481
- uncostedRecords,
482
- totalInputTokens,
483
- totalOutputTokens,
484
- totalCostUsd,
485
- verdict,
486
- diagnosis
487
- };
488
- }
489
- function buildDiagnosis(r) {
490
- if (r.totalRecords === 0) return "no records — eval produced zero runs; backend likely failed before first turn";
491
- if (r.verdict === "stub") return [
492
- `all ${r.totalRecords} records have zero token usage — the LLM backend was never called.`,
493
- "common causes: --backend sandbox without a sandbox bridge running; stub model returning hard-coded strings;",
494
- "auth misconfigured so requests were silently dropped before the LLM. Re-run with --backend tcloud and TANGLE_API_KEY set,",
495
- "or boot the cli-bridge / sandbox before invoking the eval."
496
- ].join(" ");
497
- if (r.verdict === "mixed") {
498
- const pct = (r.stubRecords / r.totalRecords * 100).toFixed(0);
499
- return [
500
- `${r.stubRecords}/${r.totalRecords} records (${pct}%) have zero token usage — the backend partially failed.`,
501
- "common causes: rate-limit cascade (429s after the first N personas);",
502
- "transient auth expiry mid-run; provider outage. Treat the affected records as missing data, not agent failures."
503
- ].join(" ");
504
- }
505
- if (r.uncostedRecords > 0) {
506
- const pct = (r.uncostedRecords / r.totalRecords * 100).toFixed(0);
507
- return [
508
- `${r.totalRecords} records with real LLM activity (in=${r.totalInputTokens}, out=${r.totalOutputTokens} tokens).`,
509
- `${r.uncostedRecords} (${pct}%) have output tokens but costUsd=0. Two distinct roots:`,
510
- "(a) cost ledger mis-wired — no usage propagation from the runtime stream into RunRecord; or",
511
- "(b) the model is unpriced at the source (sandbox/router returned $0 despite real tokens).",
512
- "For (b), price the measured tokens against the substrate table (estimateCost) instead of leaving $0."
513
- ].join(" ");
514
- }
515
- return `${r.totalRecords} records with real LLM activity (in=${r.totalInputTokens}, out=${r.totalOutputTokens} tokens, $${r.totalCostUsd.toFixed(4)}).`;
516
- }
517
- /**
518
- * Throw BackendIntegrityError if the verdict is 'stub' — i.e. every record
519
- * shows zero LLM activity. Non-strict callers can pass `{ allowMixed: false }`
520
- * to also reject mixed verdicts (recommended for CI gates).
521
- *
522
- * Real backends pass through silently.
523
- */
524
- function assertRealBackend(records, opts = {}) {
525
- return assertBackendReport(summarizeBackendIntegrity(records), opts);
526
- }
527
- function assertBackendReport(report, opts) {
528
- const allowMixed = opts.allowMixed ?? true;
529
- if (report.verdict === "stub") throw new BackendIntegrityError(`backend-integrity: ran against a stub or unconfigured backend — ${report.diagnosis}`, report);
530
- if (!allowMixed && report.verdict === "mixed") throw new BackendIntegrityError(`backend-integrity: partial backend failure rejected — ${report.diagnosis}`, report);
531
- return report;
532
- }
533
- //#endregion
534
291
  //#region src/campaign/judge-cell.ts
535
292
  /**
536
293
  * Judge scoring for one cell, with the paid-call accounting guard: a judge
@@ -1218,156 +975,6 @@ async function runEval(opts) {
1218
975
  return runCampaign(opts);
1219
976
  }
1220
977
  //#endregion
1221
- //#region src/campaign/surface-identity.ts
1222
- const GIT_OBJECT_ID = /^(?:[a-f0-9]{40}|[a-f0-9]{64})$/;
1223
- const SHA256 = /^sha256:[a-f0-9]{64}$/;
1224
- /** Validate the immutable identity shape; the owning executor verifies the Git objects and patch. */
1225
- function assertCodeSurfaceIdentity(surface) {
1226
- if (!surface || typeof surface !== "object") throw new TypeError("CodeSurface must be an object");
1227
- const candidate = surface;
1228
- if (candidate.kind !== "code") throw new TypeError("CodeSurface.kind must be \"code\"");
1229
- if (typeof candidate.worktreeRef !== "string" || candidate.worktreeRef.trim().length === 0) throw new TypeError("CodeSurface.worktreeRef must be a non-empty locator");
1230
- if (typeof candidate.baseRef !== "string" || candidate.baseRef.trim().length === 0) throw new TypeError("CodeSurface.baseRef must be a non-empty ref label");
1231
- for (const [field, value] of [
1232
- ["baseCommit", candidate.baseCommit],
1233
- ["baseTree", candidate.baseTree],
1234
- ["candidateCommit", candidate.candidateCommit],
1235
- ["candidateTree", candidate.candidateTree]
1236
- ]) if (typeof value !== "string" || !GIT_OBJECT_ID.test(value)) throw new TypeError(`CodeSurface.${field} must be a full Git object id`);
1237
- const patch = candidate.patch;
1238
- if (!patch || typeof patch !== "object" || patch.format !== "git-diff-binary") throw new TypeError("CodeSurface.patch.format must be \"git-diff-binary\"");
1239
- if (typeof patch.sha256 !== "string" || !SHA256.test(patch.sha256)) throw new TypeError("CodeSurface.patch.sha256 must be a sha256 digest");
1240
- if (!Number.isSafeInteger(patch.byteLength) || patch.byteLength < 0) throw new TypeError("CodeSurface.patch.byteLength must be a non-negative safe integer");
1241
- }
1242
- /** Assert that a value is a valid non-empty component surface. */
1243
- function assertComponentSurface(surface) {
1244
- if (!surface || typeof surface !== "object") throw new TypeError("ComponentSurface must be an object");
1245
- const candidate = surface;
1246
- if (candidate.kind !== "components") throw new TypeError("ComponentSurface.kind must be \"components\"");
1247
- if (!candidate.components || typeof candidate.components !== "object" || Array.isArray(candidate.components)) throw new TypeError("ComponentSurface.components must be an object");
1248
- const entries = Object.entries(candidate.components);
1249
- if (entries.length === 0) throw new TypeError("ComponentSurface.components must not be empty");
1250
- for (const [name, content] of entries) {
1251
- if (!name.trim() || name.trim() !== name) throw new TypeError("ComponentSurface component names must be trimmed and non-empty");
1252
- if (typeof content !== "string") throw new TypeError(`ComponentSurface component '${name}' must be a string`);
1253
- }
1254
- }
1255
- /**
1256
- * Deterministic identity material for a component surface.
1257
- *
1258
- * `canonicalString` orders keys by UTF-16 code unit (RFC 8785), which is a
1259
- * property of the value alone. The previous material ordered them with
1260
- * `localeCompare`, which reads the host's collation — so the same surface
1261
- * could produce two different identities on two machines, and the stored
1262
- * identity would stop matching a recomputation of the identical surface.
1263
- */
1264
- function componentSurfaceIdentityMaterial(surface) {
1265
- assertComponentSurface(surface);
1266
- return canonicalString({
1267
- schema: "tangle.component-surface",
1268
- components: surface.components
1269
- });
1270
- }
1271
- /**
1272
- * The retired material builder, kept PRIVATE and reachable only from
1273
- * {@link surfaceHashMatches}.
1274
- *
1275
- * A surface identity recorded before this release was minted from these bytes.
1276
- * The verify path tries the current material first and falls back to this one,
1277
- * so a stored identity still matches its own surface; nothing mints from it.
1278
- *
1279
- * Every component surface's identity moves, not only one whose names sort
1280
- * differently under the host's collation: RFC 8785 also orders the two
1281
- * top-level keys, so `components` precedes `schema` where this builder emitted
1282
- * them in literal order. The retention window therefore covers every stored
1283
- * component-surface identity, which is why this builder is kept rather than
1284
- * scoped to the mixed-case case.
1285
- */
1286
- function retiredComponentSurfaceIdentityMaterial(surface) {
1287
- assertComponentSurface(surface);
1288
- return JSON.stringify({
1289
- schema: "tangle.component-surface",
1290
- components: Object.fromEntries(Object.entries(surface.components).sort(([left], [right]) => left.localeCompare(right)))
1291
- });
1292
- }
1293
- /** Canonical, location-independent identity of a finalized code candidate.
1294
- * Commit metadata is excluded: two commits with the same base, final tree,
1295
- * and patch bytes are the same executable candidate. */
1296
- function codeSurfaceIdentityMaterial(surface) {
1297
- assertCodeSurfaceIdentity(surface);
1298
- return JSON.stringify({
1299
- schema: "tangle.code-surface",
1300
- baseCommit: surface.baseCommit,
1301
- baseTree: surface.baseTree,
1302
- candidateTree: surface.candidateTree,
1303
- patch: {
1304
- format: surface.patch.format,
1305
- sha256: surface.patch.sha256,
1306
- byteLength: surface.patch.byteLength
1307
- }
1308
- });
1309
- }
1310
- /** Full SHA-256 content identity for a prompt or finalized code surface. */
1311
- function surfaceContentHash(surface) {
1312
- const material = typeof surface === "string" ? surface : surface.kind === "components" ? componentSurfaceIdentityMaterial(surface) : codeSurfaceIdentityMaterial(surface);
1313
- return `sha256:${createHash("sha256").update(material).digest("hex")}`;
1314
- }
1315
- /** Short loop key derived from the same content identity as provenance. */
1316
- function surfaceHash(surface) {
1317
- return surfaceContentHash(surface).slice(7, 23);
1318
- }
1319
- /**
1320
- * Whether `storedHash` is the loop key of `surface`, under the current identity
1321
- * material or the retired one.
1322
- *
1323
- * A stored key is 16 hex characters with no room for a scheme tag, so the
1324
- * scheme cannot be read off the value the way an `agent-profile-cell` id names
1325
- * its own. The verify path therefore tries both, which gives the same property:
1326
- * a key minted by an earlier release still matches its own surface, and a
1327
- * surface that was actually edited matches neither.
1328
- *
1329
- * Only a component surface can differ between the two; a prompt or code surface
1330
- * produces identical material under both, so the second comparison is a no-op
1331
- * for them.
1332
- */
1333
- function surfaceHashMatches(surface, storedHash) {
1334
- if (surfaceHash(surface) === storedHash) return true;
1335
- if (typeof surface === "string" || surface.kind !== "components") return false;
1336
- return createHash("sha256").update(retiredComponentSurfaceIdentityMaterial(surface)).digest("hex").slice(0, 16) === storedHash;
1337
- }
1338
- /** Canonical customer-visible description of the exact before/after surfaces. */
1339
- function renderSurfaceDiff(winnerSurface, baselineSurface) {
1340
- if (typeof winnerSurface === "string" && typeof baselineSurface === "string") return [
1341
- "--- baseline",
1342
- "+++ winner",
1343
- ...baselineSurface.split("\n").map((line) => `- ${line}`),
1344
- ...winnerSurface.split("\n").map((line) => `+ ${line}`)
1345
- ].join("\n");
1346
- const describe = (surface) => {
1347
- if (typeof surface === "string") return "(prompt surface)";
1348
- if (surface.kind === "components") {
1349
- assertComponentSurface(surface);
1350
- return Object.entries(surface.components).sort(([left], [right]) => left.localeCompare(right)).map(([name, content]) => `[${name}]\n${content}`).join("\n\n");
1351
- }
1352
- assertCodeSurfaceIdentity(surface);
1353
- return [
1354
- `baseCommit=${surface.baseCommit}`,
1355
- `baseTree=${surface.baseTree}`,
1356
- `candidateCommit=${surface.candidateCommit}`,
1357
- `candidateTree=${surface.candidateTree}`,
1358
- `patch=${surface.patch.sha256}`,
1359
- `patchBytes=${surface.patch.byteLength}`,
1360
- ...surface.summary ? [surface.summary] : []
1361
- ].join("\n");
1362
- };
1363
- return `--- baseline\n${describe(baselineSurface)}\n+++ winner\n${describe(winnerSurface)}`;
1364
- }
1365
- /** Bind a campaign cache entry to the exact surface and caller-owned execution revision. */
1366
- function surfaceDispatchRef(surface, executionRef = "anonymous") {
1367
- if (!executionRef.trim() || executionRef.trim() !== executionRef) throw new Error("surfaceDispatchRef: executionRef must be trimmed and non-empty");
1368
- return `surface:${executionRef}:${surfaceContentHash(surface)}`;
1369
- }
1370
- //#endregion
1371
978
  //#region src/canary.ts
1372
979
  /**
1373
980
  * Run all configured canaries against a chronological run list.
@@ -2792,403 +2399,767 @@ function paretoFrontierWithCrowding(candidates, objectives) {
2792
2399
  return crowdingDistance(frontier, objectives).sort((a, b) => b.distance - a.distance);
2793
2400
  }
2794
2401
  //#endregion
2795
- //#region src/json-recovery.ts
2402
+ //#region src/campaign/search-ledger.ts
2796
2403
  /**
2797
- * Truncation-tolerant JSON recovery shared by every parser that reads JSON
2798
- * out of a model response (reflective-mutation proposals, judge scores, the
2799
- * completion-correctness checker).
2404
+ * Durable append-only audit log for improvement searches.
2800
2405
  *
2801
- * LLMs routinely hit a max_tokens cap mid-emission, leaving a JSON prefix
2802
- * with an unclosed string / object / array and often a dangling key or
2803
- * trailing comma. Throwing on that prefix and letting the throw fold into
2804
- * a fabricated zero score downstream is the bug class this module exists
2805
- * to prevent (see `JudgeParseError`'s contract: a synthetic zero is
2806
- * indistinguishable from a real low score). Recovering the complete prefix
2807
- * turns a would-be fabricated zero into a real measurement.
2808
- */
2809
- /**
2810
- * Walk the input as JSON-aware (string vs not, escape-aware) and close
2811
- * unclosed `{` / `[` in LIFO order at the tail. If the input was already
2812
- * balanced returns it unchanged. If a string was open at end-of-input we
2813
- * also close it with `"` first, since a truncated string-mid-value is the
2814
- * most common LLM cap-hit failure mode and JSON.parse cannot proceed
2815
- * without one.
2406
+ * Existing campaign artifacts keep their own rich records: `RunRecord` owns a
2407
+ * measured run and `CostLedger` owns per-call accounting. This ledger does not
2408
+ * copy those structures. It binds their immutable ids and receipts into one replayable event stream so a
2409
+ * search can answer, after a crash, exactly which candidates and task attempts
2410
+ * existed, which surfaces actually fired, what they cost, and why they were
2411
+ * selected or rejected.
2412
+ *
2413
+ * The file format is canonical JSONL with a SHA-256 hash chain. Every append is
2414
+ * serialized across processes, fsynced before acknowledgement, and idempotent
2415
+ * by `eventId`. A malformed, non-canonical, truncated, reordered, or conflicting
2416
+ * log fails loudly; the implementation never skips a bad row.
2816
2417
  *
2817
- * Returns null when the structure is unrecoverable (e.g. depth would go
2818
- * negative that's an *over*-closed prefix, not a truncation).
2418
+ * The journal machinery itself (hash chain, locking, fsync, idempotent append)
2419
+ * is the generic `ledger-core` journal; this module supplies the campaign
2420
+ * codec: event schemas, canonical event ordering, and the search state machine.
2819
2421
  */
2820
- function autoCloseTruncatedJson(raw) {
2821
- const stack = [];
2822
- let inString = false;
2823
- let escaped = false;
2824
- for (const c of raw) {
2825
- if (escaped) {
2826
- escaped = false;
2827
- continue;
2422
+ const SEARCH_LEDGER_SCHEMA = "tangle.search-ledger.v1";
2423
+ const NON_EMPTY = z.string().min(1).refine((value) => value.trim() === value, "must not contain surrounding whitespace");
2424
+ const HASH = z.string().regex(/^sha256:[a-f0-9]{64}$/);
2425
+ const LINEAGE_NODE_ID = z.string().regex(/^[a-f0-9]{16}$/);
2426
+ const IMMUTABLE_REVISION = z.string().regex(/^(?:[a-f0-9]{40}|[a-f0-9]{64}|sha256:[a-f0-9]{64}|sha512:[A-Za-z0-9+/=]+)$/);
2427
+ const ISO_TIMESTAMP = z.string().regex(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/).refine((value) => Number.isFinite(Date.parse(value)), "invalid timestamp");
2428
+ const NON_NEGATIVE_INT = z.number().int().nonnegative().safe();
2429
+ const FINITE_NUMBER = z.number().finite();
2430
+ const ArtifactRefSchema = z.object({
2431
+ role: NON_EMPTY,
2432
+ uri: NON_EMPTY,
2433
+ sha256: HASH,
2434
+ byteLength: NON_NEGATIVE_INT
2435
+ }).strict();
2436
+ const SourceRefSchema = z.object({
2437
+ uri: NON_EMPTY,
2438
+ revision: IMMUTABLE_REVISION
2439
+ }).strict();
2440
+ const FailureReasonSchema = z.object({
2441
+ code: NON_EMPTY,
2442
+ message: NON_EMPTY
2443
+ }).strict();
2444
+ const EventBaseShape = {
2445
+ eventId: NON_EMPTY,
2446
+ occurredAt: ISO_TIMESTAMP,
2447
+ artifacts: z.array(ArtifactRefSchema).min(1)
2448
+ };
2449
+ const OperationKindSchema = z.enum([
2450
+ "candidate-generation",
2451
+ "analysis",
2452
+ "selection",
2453
+ "judge",
2454
+ "other"
2455
+ ]);
2456
+ const CandidateSlotSchema = z.object({
2457
+ slotId: NON_EMPTY,
2458
+ generationOperationId: NON_EMPTY
2459
+ }).strict();
2460
+ const PlannedOperationSchema = z.object({
2461
+ operationId: NON_EMPTY,
2462
+ kind: OperationKindSchema
2463
+ }).strict();
2464
+ const SearchPlanExtendedSchema = z.object({
2465
+ ...EventBaseShape,
2466
+ kind: z.literal("search-plan-extended"),
2467
+ extension: z.object({
2468
+ candidateSlots: z.array(CandidateSlotSchema),
2469
+ operations: z.array(PlannedOperationSchema)
2470
+ }).strict().superRefine((extension, ctx) => {
2471
+ if (extension.candidateSlots.length === 0 && extension.operations.length === 0) ctx.addIssue({
2472
+ code: "custom",
2473
+ message: "a plan extension must add slots or operations"
2474
+ });
2475
+ })
2476
+ }).strict();
2477
+ const SearchPlannedSchema = z.object({
2478
+ ...EventBaseShape,
2479
+ kind: z.literal("search-planned"),
2480
+ plan: z.object({
2481
+ candidateSlots: z.array(CandidateSlotSchema).min(1),
2482
+ tasks: z.array(z.object({
2483
+ taskId: NON_EMPTY,
2484
+ source: SourceRefSchema,
2485
+ benchmark: SourceRefSchema,
2486
+ maxAttempts: z.number().int().positive().safe()
2487
+ }).strict()).min(1),
2488
+ operations: z.array(PlannedOperationSchema).min(1)
2489
+ }).strict()
2490
+ }).strict();
2491
+ const CandidateRegisteredSchema = z.object({
2492
+ ...EventBaseShape,
2493
+ kind: z.literal("candidate-registered"),
2494
+ slotId: NON_EMPTY,
2495
+ generationOperationId: NON_EMPTY,
2496
+ candidateId: NON_EMPTY,
2497
+ lineage: z.object({
2498
+ lineageNodeId: LINEAGE_NODE_ID,
2499
+ parentCandidateIds: z.array(NON_EMPTY),
2500
+ generation: NON_NEGATIVE_INT,
2501
+ proposer: NON_EMPTY,
2502
+ proposerSource: SourceRefSchema
2503
+ }).strict(),
2504
+ surfaces: z.array(z.object({
2505
+ surfaceId: NON_EMPTY,
2506
+ kind: z.enum([
2507
+ "prompt",
2508
+ "tool-contract",
2509
+ "runtime-config",
2510
+ "memory",
2511
+ "knowledge",
2512
+ "agent-profile",
2513
+ "code",
2514
+ "deployment"
2515
+ ]),
2516
+ artifact: ArtifactRefSchema
2517
+ }).strict()).min(1)
2518
+ }).strict();
2519
+ const CandidateSlotClosedSchema = z.object({
2520
+ ...EventBaseShape,
2521
+ kind: z.literal("candidate-slot-closed"),
2522
+ slotId: NON_EMPTY,
2523
+ generationOperationId: NON_EMPTY,
2524
+ reason: FailureReasonSchema
2525
+ }).strict();
2526
+ const KnownTokensSchema = z.object({
2527
+ status: z.literal("known"),
2528
+ inputTokens: NON_NEGATIVE_INT,
2529
+ outputTokens: NON_NEGATIVE_INT,
2530
+ cachedTokens: NON_NEGATIVE_INT
2531
+ }).strict();
2532
+ const UnknownSchema = z.object({
2533
+ status: z.literal("unknown"),
2534
+ reason: NON_EMPTY
2535
+ }).strict();
2536
+ const KnownCostSchema = z.object({
2537
+ status: z.literal("known"),
2538
+ usd: z.number().finite().nonnegative(),
2539
+ source: z.enum([
2540
+ "provider",
2541
+ "pricing-table",
2542
+ "free"
2543
+ ])
2544
+ }).strict().superRefine((cost, ctx) => {
2545
+ if (cost.source === "free" && cost.usd !== 0) ctx.addIssue({
2546
+ code: "custom",
2547
+ message: "free cost source must have usd 0"
2548
+ });
2549
+ });
2550
+ const UnknownCostSchema = z.object({
2551
+ status: z.literal("unknown"),
2552
+ knownLowerBoundUsd: z.number().finite().nonnegative(),
2553
+ reason: NON_EMPTY
2554
+ }).strict();
2555
+ const AccountingSchema = z.object({
2556
+ tokens: z.discriminatedUnion("status", [KnownTokensSchema, UnknownSchema]),
2557
+ cost: z.discriminatedUnion("status", [KnownCostSchema, UnknownCostSchema])
2558
+ }).strict();
2559
+ const MetricsSchema = z.record(NON_EMPTY, FINITE_NUMBER).superRefine((metrics, ctx) => {
2560
+ for (const key of Object.keys(metrics)) if (key === "__proto__" || key === "constructor" || key === "prototype") ctx.addIssue({
2561
+ code: "custom",
2562
+ message: `unsafe metric key ${key}`
2563
+ });
2564
+ });
2565
+ const OutcomeSchema = z.discriminatedUnion("status", [
2566
+ z.object({
2567
+ status: z.literal("passed"),
2568
+ score: FINITE_NUMBER,
2569
+ metrics: MetricsSchema
2570
+ }).strict(),
2571
+ z.object({
2572
+ status: z.literal("failed"),
2573
+ score: FINITE_NUMBER,
2574
+ metrics: MetricsSchema,
2575
+ failure: FailureReasonSchema
2576
+ }).strict(),
2577
+ z.object({
2578
+ status: z.literal("errored"),
2579
+ metrics: MetricsSchema,
2580
+ error: FailureReasonSchema.extend({ retryable: z.boolean() }).strict()
2581
+ }).strict()
2582
+ ]);
2583
+ const EffectSchema = z.discriminatedUnion("status", [z.object({
2584
+ status: z.literal("measured"),
2585
+ metric: NON_EMPTY,
2586
+ baselineValue: FINITE_NUMBER,
2587
+ candidateValue: FINITE_NUMBER,
2588
+ delta: FINITE_NUMBER
2589
+ }).strict().superRefine((effect, ctx) => {
2590
+ const expected = effect.candidateValue - effect.baselineValue;
2591
+ const tolerance = Number.EPSILON * Math.max(1, Math.abs(expected), Math.abs(effect.delta)) * 8;
2592
+ if (Math.abs(effect.delta - expected) > tolerance) ctx.addIssue({
2593
+ code: "custom",
2594
+ message: "delta must equal candidateValue - baselineValue"
2595
+ });
2596
+ }), z.object({
2597
+ status: z.literal("not-measured"),
2598
+ reason: NON_EMPTY
2599
+ }).strict()]);
2600
+ const SurfaceEvidenceSchema = z.object({
2601
+ surfaceId: NON_EMPTY,
2602
+ fired: z.boolean(),
2603
+ firingCount: NON_NEGATIVE_INT,
2604
+ effect: EffectSchema,
2605
+ evidence: z.array(ArtifactRefSchema).min(1)
2606
+ }).strict().superRefine((evidence, ctx) => {
2607
+ if (evidence.fired && evidence.firingCount === 0) ctx.addIssue({
2608
+ code: "custom",
2609
+ message: "a fired surface must have firingCount >= 1"
2610
+ });
2611
+ if (!evidence.fired && evidence.firingCount !== 0) ctx.addIssue({
2612
+ code: "custom",
2613
+ message: "a surface that did not fire must have firingCount 0"
2614
+ });
2615
+ if (!evidence.fired && evidence.effect.status === "measured" && evidence.effect.delta !== 0) ctx.addIssue({
2616
+ code: "custom",
2617
+ message: "a surface that did not fire cannot claim non-zero effect"
2618
+ });
2619
+ });
2620
+ const TaskAttemptedSchema = z.object({
2621
+ ...EventBaseShape,
2622
+ kind: z.literal("task-attempted"),
2623
+ candidateId: NON_EMPTY,
2624
+ runId: NON_EMPTY,
2625
+ attemptIndex: NON_NEGATIVE_INT,
2626
+ task: z.object({
2627
+ taskId: NON_EMPTY,
2628
+ source: SourceRefSchema
2629
+ }).strict(),
2630
+ identity: z.object({
2631
+ model: z.object({
2632
+ provider: NON_EMPTY,
2633
+ snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
2634
+ }).strict(),
2635
+ agent: SourceRefSchema,
2636
+ benchmark: SourceRefSchema
2637
+ }).strict(),
2638
+ outcome: OutcomeSchema,
2639
+ accounting: AccountingSchema,
2640
+ surfaceEvidence: z.array(SurfaceEvidenceSchema).min(1)
2641
+ }).strict();
2642
+ const SearchOperationRecordedSchema = z.object({
2643
+ ...EventBaseShape,
2644
+ kind: z.literal("search-operation-recorded"),
2645
+ operationId: NON_EMPTY,
2646
+ operationKind: OperationKindSchema,
2647
+ execution: z.discriminatedUnion("kind", [z.object({
2648
+ kind: z.literal("model"),
2649
+ model: z.object({
2650
+ provider: NON_EMPTY,
2651
+ snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
2652
+ }).strict(),
2653
+ source: SourceRefSchema
2654
+ }).strict(), z.object({
2655
+ kind: z.literal("deterministic"),
2656
+ source: SourceRefSchema
2657
+ }).strict()]),
2658
+ outcome: z.discriminatedUnion("status", [
2659
+ z.object({ status: z.literal("completed") }).strict(),
2660
+ z.object({
2661
+ status: z.literal("partial"),
2662
+ failure: FailureReasonSchema
2663
+ }).strict(),
2664
+ z.object({
2665
+ status: z.literal("failed"),
2666
+ failure: FailureReasonSchema
2667
+ }).strict()
2668
+ ]),
2669
+ accounting: AccountingSchema
2670
+ }).strict();
2671
+ const CandidateDecidedSchema = z.object({
2672
+ ...EventBaseShape,
2673
+ kind: z.literal("candidate-decided"),
2674
+ candidateId: NON_EMPTY,
2675
+ decision: z.discriminatedUnion("status", [z.object({ status: z.literal("selected") }).strict(), z.object({
2676
+ status: z.literal("rejected"),
2677
+ reason: FailureReasonSchema
2678
+ }).strict()])
2679
+ }).strict();
2680
+ const SearchCompletedSchema = z.object({
2681
+ ...EventBaseShape,
2682
+ kind: z.literal("search-completed"),
2683
+ result: z.discriminatedUnion("status", [z.object({
2684
+ status: z.literal("selected"),
2685
+ candidateId: NON_EMPTY
2686
+ }).strict(), z.object({
2687
+ status: z.literal("all-rejected"),
2688
+ reason: FailureReasonSchema
2689
+ }).strict()])
2690
+ }).strict();
2691
+ const EventSchema = z.discriminatedUnion("kind", [
2692
+ SearchPlannedSchema,
2693
+ SearchPlanExtendedSchema,
2694
+ CandidateRegisteredSchema,
2695
+ CandidateSlotClosedSchema,
2696
+ TaskAttemptedSchema,
2697
+ SearchOperationRecordedSchema,
2698
+ CandidateDecidedSchema,
2699
+ SearchCompletedSchema
2700
+ ]);
2701
+ const EntrySchema = z.object({
2702
+ schema: z.literal(SEARCH_LEDGER_SCHEMA),
2703
+ campaignId: NON_EMPTY,
2704
+ sequence: NON_NEGATIVE_INT,
2705
+ previousHash: z.union([HASH, z.null()]),
2706
+ event: EventSchema,
2707
+ entryHash: HASH
2708
+ }).strict();
2709
+ /** Validate and return a canonical copy. Arrays whose order is not semantic are
2710
+ * sorted so retries from different processes produce byte-identical events. */
2711
+ function validateSearchLedgerEvent(input) {
2712
+ const parsed = EventSchema.safeParse(input);
2713
+ if (!parsed.success) throw new SearchLedgerError(`invalid search ledger event: ${formatZodError(parsed.error)}`);
2714
+ return normalizeEvent(parsed.data);
2715
+ }
2716
+ /** Open a durable filesystem search ledger. Construction performs no I/O; the
2717
+ * first `append` or `replay` validates the complete existing file. */
2718
+ function openSearchLedger(options) {
2719
+ if (options.path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
2720
+ return new FileSearchLedger(options.path, options.campaignId, options.trustedHead);
2721
+ }
2722
+ /** Replay immutable search-ledger JSONL through the same codec as FileSearchLedger. */
2723
+ function replaySearchLedgerText(text, campaignId, source) {
2724
+ return replayLedgerText(text, source, searchLedgerCodec(campaignId)).projection;
2725
+ }
2726
+ function searchLedgerCodec(campaignId) {
2727
+ return {
2728
+ ...SEARCH_LEDGER_FILE_CONTEXT,
2729
+ header: {
2730
+ schema: SEARCH_LEDGER_SCHEMA,
2731
+ campaignId
2732
+ },
2733
+ conflictError: (message) => new SearchLedgerConflictError(message),
2734
+ parseEntry: parseSearchLedgerEntry,
2735
+ checkEntryHeader: (entry, index) => {
2736
+ if (entry.campaignId !== campaignId) throw new SearchLedgerIntegrityError(`entry ${index} belongs to campaign ${entry.campaignId}, expected ${campaignId}`);
2737
+ },
2738
+ createProjector: () => createSearchLedgerProjector(campaignId)
2739
+ };
2740
+ }
2741
+ /** Append-only file-backed search ledger with idempotent writes and replay. */
2742
+ var FileSearchLedger = class {
2743
+ path;
2744
+ campaignId;
2745
+ trustedHeadPath;
2746
+ trustedHeadMode;
2747
+ journal;
2748
+ constructor(path, campaignId, trustedHead = "pin") {
2749
+ if (path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
2750
+ if (campaignId.length === 0) throw new SearchLedgerError("campaignId is empty");
2751
+ if (campaignId.trim() !== campaignId) throw new SearchLedgerError("campaignId must not contain surrounding whitespace");
2752
+ this.campaignId = campaignId;
2753
+ this.trustedHeadMode = trustedHead;
2754
+ this.journal = new FileLedgerJournal(path, searchLedgerCodec(campaignId), { requireTrustedHead: trustedHead === "require" });
2755
+ this.path = this.journal.path;
2756
+ this.trustedHeadPath = this.journal.trustedHeadPath;
2757
+ }
2758
+ async replay() {
2759
+ return (await this.journal.replay()).projection;
2760
+ }
2761
+ async append(input) {
2762
+ const event = validateSearchLedgerEvent(input);
2763
+ const { entry, appended, projection } = await this.journal.append(event, { pinHead: this.trustedHeadMode !== "off" });
2764
+ return {
2765
+ entry,
2766
+ appended,
2767
+ replay: projection
2768
+ };
2769
+ }
2770
+ async trustedHead() {
2771
+ return this.journal.trustedHead();
2772
+ }
2773
+ async pinTrustedHead() {
2774
+ return this.journal.pinTrustedHead();
2775
+ }
2776
+ async clearTrustedHead() {
2777
+ return this.journal.clearTrustedHead();
2778
+ }
2779
+ };
2780
+ function parseSearchLedgerEntry(raw, context) {
2781
+ const parsed = EntrySchema.safeParse(raw);
2782
+ if (!parsed.success) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has a malformed entry at line ${context.line}: ${formatZodError(parsed.error)}`);
2783
+ const entry = parsed.data;
2784
+ if (canonicalString(validateSearchLedgerEvent(entry.event)) !== canonicalString(entry.event)) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has non-canonical event ordering at line ${context.line}`);
2785
+ return entry;
2786
+ }
2787
+ /** Replay the campaign search state machine over chain-verified entries. The
2788
+ * generic journal owns sequence, hash, and eventId-uniqueness checks; this
2789
+ * projector owns every campaign invariant and builds the replay projection. */
2790
+ function createSearchLedgerProjector(campaignId) {
2791
+ const candidates = /* @__PURE__ */ new Map();
2792
+ const plannedSlots = /* @__PURE__ */ new Map();
2793
+ const plannedOperations = /* @__PURE__ */ new Map();
2794
+ const planExtensions = [];
2795
+ const candidateBySlot = /* @__PURE__ */ new Map();
2796
+ const closedSlots = /* @__PURE__ */ new Map();
2797
+ const lineageNodes = /* @__PURE__ */ new Map();
2798
+ const runIds = /* @__PURE__ */ new Set();
2799
+ const attemptKeys = /* @__PURE__ */ new Set();
2800
+ const candidateEvents = [];
2801
+ const closedSlotEvents = [];
2802
+ const attempts = [];
2803
+ const operationEvents = [];
2804
+ const operationsById = /* @__PURE__ */ new Map();
2805
+ const decisions = [];
2806
+ let planEvent = null;
2807
+ let completion = null;
2808
+ let previousOccurredAt = Number.NEGATIVE_INFINITY;
2809
+ const apply = (entry, index) => {
2810
+ const event = entry.event;
2811
+ if (completion) throw new SearchLedgerIntegrityError(`event ${event.eventId} appears after terminal event ${completion.eventId}`);
2812
+ const occurredAt = Date.parse(event.occurredAt);
2813
+ if (occurredAt < previousOccurredAt) throw new SearchLedgerIntegrityError(`event ${event.eventId} occurred before the preceding durable event`);
2814
+ previousOccurredAt = occurredAt;
2815
+ assertUnique(event.artifacts.map(artifactKey), "artifact receipt", event.eventId);
2816
+ if (event.kind === "search-planned") {
2817
+ if (index !== 0 || planEvent) throw new SearchLedgerIntegrityError("search plan must be the first and only plan event");
2818
+ assertUnique(event.plan.candidateSlots.map((slot) => slot.slotId), "candidate slot", event.eventId);
2819
+ assertUnique(event.plan.tasks.map((task) => task.taskId), "planned taskId", event.eventId);
2820
+ assertUnique(event.plan.operations.map((operation) => operation.operationId), "planned operationId", event.eventId);
2821
+ for (const operation of event.plan.operations) plannedOperations.set(operation.operationId, operation);
2822
+ for (const slot of event.plan.candidateSlots) {
2823
+ assertSlotGenerationOperation(slot, plannedOperations);
2824
+ plannedSlots.set(slot.slotId, slot);
2825
+ }
2826
+ planEvent = event;
2827
+ return;
2828
2828
  }
2829
- if (inString) {
2830
- if (c === "\\") {
2831
- escaped = true;
2832
- continue;
2829
+ if (!planEvent) throw new SearchLedgerIntegrityError(`event ${event.eventId} appears before the required search plan`);
2830
+ if (event.kind === "search-plan-extended") {
2831
+ assertUnique(event.extension.candidateSlots.map((slot) => slot.slotId), "candidate slot", event.eventId);
2832
+ assertUnique(event.extension.operations.map((operation) => operation.operationId), "planned operationId", event.eventId);
2833
+ for (const operation of event.extension.operations) {
2834
+ if (plannedOperations.has(operation.operationId)) throw new SearchLedgerIntegrityError(`plan extension ${event.eventId} re-plans operation ${operation.operationId}`);
2835
+ plannedOperations.set(operation.operationId, operation);
2833
2836
  }
2834
- if (c === "\"") {
2835
- inString = false;
2836
- continue;
2837
+ for (const slot of event.extension.candidateSlots) {
2838
+ if (plannedSlots.has(slot.slotId)) throw new SearchLedgerIntegrityError(`plan extension ${event.eventId} re-plans candidate slot ${slot.slotId}`);
2839
+ assertSlotGenerationOperation(slot, plannedOperations);
2840
+ plannedSlots.set(slot.slotId, slot);
2837
2841
  }
2838
- continue;
2839
- }
2840
- if (c === "\"") {
2841
- inString = true;
2842
- continue;
2842
+ planExtensions.push(event);
2843
+ return;
2843
2844
  }
2844
- if (c === "{" || c === "[") stack.push(c);
2845
- else if (c === "}") {
2846
- if (stack.pop() !== "{") return null;
2847
- } else if (c === "]") {
2848
- if (stack.pop() !== "[") return null;
2845
+ if (event.kind === "candidate-registered") {
2846
+ if (candidates.has(event.candidateId)) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} was registered twice`);
2847
+ const plannedSlot = plannedSlots.get(event.slotId);
2848
+ if (!plannedSlot) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} binds unknown slot ${event.slotId}`);
2849
+ if (candidateBySlot.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was bound twice`);
2850
+ if (closedSlots.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was already closed`);
2851
+ if (event.generationOperationId !== plannedSlot.generationOperationId) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} generation operation ${event.generationOperationId} does not match slot ${event.slotId} plan ${plannedSlot.generationOperationId}`);
2852
+ const generationOperation = operationsById.get(event.generationOperationId);
2853
+ if (!generationOperation) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} precedes generation operation ${event.generationOperationId}`);
2854
+ if (generationOperation.outcome.status === "failed") throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} cannot bind failed generation operation ${event.generationOperationId}`);
2855
+ const previousCandidate = lineageNodes.get(event.lineage.lineageNodeId);
2856
+ if (previousCandidate) throw new SearchLedgerIntegrityError(`lineage node ${event.lineage.lineageNodeId} is already bound to ${previousCandidate}`);
2857
+ assertUnique(event.lineage.parentCandidateIds, "parentCandidateId", event.eventId);
2858
+ const parents = event.lineage.parentCandidateIds.map((id) => {
2859
+ const parent = candidates.get(id);
2860
+ if (!parent) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} references unknown parent ${id}`);
2861
+ return parent;
2862
+ });
2863
+ const expectedGeneration = parents.length === 0 ? 0 : Math.max(...parents.map((parent) => parent.registered.lineage.generation)) + 1;
2864
+ if (event.lineage.generation !== expectedGeneration) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} generation ${event.lineage.generation} does not follow its parents (expected ${expectedGeneration})`);
2865
+ assertUnique(event.surfaces.map((surface) => surface.surfaceId), "surfaceId", event.eventId);
2866
+ candidates.set(event.candidateId, {
2867
+ registered: event,
2868
+ attempts: [],
2869
+ decision: null
2870
+ });
2871
+ candidateBySlot.set(event.slotId, event.candidateId);
2872
+ lineageNodes.set(event.lineage.lineageNodeId, event.candidateId);
2873
+ candidateEvents.push(event);
2874
+ return;
2849
2875
  }
2850
- }
2851
- if (stack.length === 0 && !inString) return raw;
2852
- let suffix = "";
2853
- if (escaped) suffix += "\\";
2854
- if (inString) suffix += "\"";
2855
- while (stack.length > 0) {
2856
- const opener = stack.pop();
2857
- suffix += opener === "{" ? "}" : "]";
2858
- }
2859
- return raw + suffix;
2860
- }
2861
- const UNPARSEABLE = Symbol("unparseable");
2862
- function tryParse(candidate) {
2863
- try {
2864
- return JSON.parse(candidate);
2865
- } catch {
2866
- return UNPARSEABLE;
2867
- }
2868
- }
2869
- /** Index of the last `,` that sits outside any string literal, or -1. Cutting
2870
- * there discards a dangling key / half-emitted value at the tail while
2871
- * keeping every complete member before it. */
2872
- function lastCommaOutsideString(s) {
2873
- let inString = false;
2874
- let escaped = false;
2875
- let last = -1;
2876
- for (let i = 0; i < s.length; i++) {
2877
- const c = s[i];
2878
- if (escaped) {
2879
- escaped = false;
2880
- continue;
2876
+ if (event.kind === "task-attempted") {
2877
+ const candidate = candidates.get(event.candidateId);
2878
+ if (!candidate) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} references unknown candidate ${event.candidateId}`);
2879
+ if (candidate.decision) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} appears after candidate ${event.candidateId} was decided`);
2880
+ const plannedTask = planEvent.plan.tasks.find((task) => task.taskId === event.task.taskId);
2881
+ if (!plannedTask) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} references unplanned task ${event.task.taskId}`);
2882
+ if (canonicalString(plannedTask.source) !== canonicalString(event.task.source) || canonicalString(plannedTask.benchmark) !== canonicalString(event.identity.benchmark)) throw new SearchLedgerIntegrityError(`task ${event.task.taskId} does not match its planned source identity`);
2883
+ if (event.attemptIndex >= plannedTask.maxAttempts) throw new SearchLedgerIntegrityError(`task ${event.task.taskId} attempt ${event.attemptIndex} exceeds planned maxAttempts ${plannedTask.maxAttempts}`);
2884
+ if (runIds.has(event.runId)) throw new SearchLedgerIntegrityError(`runId ${event.runId} was recorded twice`);
2885
+ runIds.add(event.runId);
2886
+ const attemptKey = canonicalString([
2887
+ event.candidateId,
2888
+ event.task.taskId,
2889
+ event.attemptIndex
2890
+ ]);
2891
+ if (attemptKeys.has(attemptKey)) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} attempt ${event.attemptIndex} was recorded twice`);
2892
+ const expectedAttemptIndex = candidate.attempts.filter((attempt) => attempt.task.taskId === event.task.taskId).length;
2893
+ if (event.attemptIndex !== expectedAttemptIndex) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} attempt index ${event.attemptIndex} is not contiguous (expected ${expectedAttemptIndex})`);
2894
+ const previousAttempt = candidate.attempts.find((attempt) => attempt.task.taskId === event.task.taskId);
2895
+ if (previousAttempt?.outcome.status !== void 0 && previousAttempt.outcome.status !== "errored") throw new SearchLedgerIntegrityError(`task ${event.task.taskId} was retried after a measured outcome`);
2896
+ if (previousAttempt && canonicalString({
2897
+ task: previousAttempt.task,
2898
+ identity: previousAttempt.identity
2899
+ }) !== canonicalString({
2900
+ task: event.task,
2901
+ identity: event.identity
2902
+ })) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} task ${event.task.taskId} changed immutable execution identity between attempts`);
2903
+ attemptKeys.add(attemptKey);
2904
+ const declared = candidate.registered.surfaces.map((surface) => surface.surfaceId).sort();
2905
+ const observed = event.surfaceEvidence.map((surface) => surface.surfaceId).sort();
2906
+ assertUnique(observed, "surface evidence", event.eventId);
2907
+ for (const evidence of event.surfaceEvidence) assertUnique(evidence.evidence.map(artifactKey), "surface evidence receipt", event.eventId);
2908
+ if (canonicalString(declared) !== canonicalString(observed)) throw new SearchLedgerIntegrityError(`attempt ${event.eventId} surface evidence does not exactly cover candidate ${event.candidateId}`);
2909
+ candidate.attempts.push(event);
2910
+ attempts.push(event);
2911
+ return;
2881
2912
  }
2882
- if (inString) {
2883
- if (c === "\\") escaped = true;
2884
- else if (c === "\"") inString = false;
2885
- continue;
2913
+ if (event.kind === "search-operation-recorded") {
2914
+ const plannedOperation = plannedOperations.get(event.operationId);
2915
+ if (!plannedOperation) throw new SearchLedgerIntegrityError(`operation ${event.operationId} was not declared in the search plan`);
2916
+ if (plannedOperation.kind !== event.operationKind) throw new SearchLedgerIntegrityError(`operation ${event.operationId} kind ${event.operationKind} does not match planned ${plannedOperation.kind}`);
2917
+ if (operationsById.has(event.operationId)) throw new SearchLedgerIntegrityError(`operation ${event.operationId} was recorded twice`);
2918
+ operationsById.set(event.operationId, event);
2919
+ operationEvents.push(event);
2920
+ return;
2886
2921
  }
2887
- if (c === "\"") inString = true;
2888
- else if (c === ",") last = i;
2889
- }
2890
- return last;
2891
- }
2892
- /**
2893
- * Best-effort parse of a possibly-truncated JSON payload embedded in model
2894
- * output (prose and markdown fences tolerated). Tries each opener position
2895
- * (earliest `{`/`[` first, then the other when the first yields nothing
2896
- * prose like `[note] {"score":3}` must not poison the slice), and from each
2897
- * start, in order:
2898
- *
2899
- * 1. plain `JSON.parse` of the opener → last-closer slice,
2900
- * 2. auto-closing unclosed structures at the tail
2901
- * (`autoCloseTruncatedJson`),
2902
- * 3. trimming the tail back to the previous complete member boundary (the
2903
- * last comma outside a string) and auto-closing again, repeatedly.
2904
- *
2905
- * Recovers e.g. `{"correct": false, "` → `{ correct: false }` — the exact
2906
- * cap-hit shape that has zeroed real eval rows. Returns the parsed value
2907
- * (always an object or array, given the slice starts at an opener), or
2908
- * `null` when nothing parseable can be recovered. Never throws.
2909
- */
2910
- function recoverTruncatedJson(text) {
2911
- const starts = [text.indexOf("{"), text.indexOf("[")].filter((i) => i >= 0).sort((a, b) => a - b);
2912
- for (const start of starts) {
2913
- const recovered = recoverFrom(text.slice(start));
2914
- if (recovered !== UNPARSEABLE) return recovered;
2915
- }
2916
- return null;
2917
- }
2918
- function recoverFrom(slice) {
2919
- let candidate = slice;
2920
- const lastClose = Math.max(candidate.lastIndexOf("}"), candidate.lastIndexOf("]"));
2921
- if (lastClose > 0) {
2922
- const balanced = tryParse(candidate.slice(0, lastClose + 1));
2923
- if (balanced !== UNPARSEABLE) return balanced;
2924
- }
2925
- for (let i = 0; i < 64; i++) {
2926
- const closed = autoCloseTruncatedJson(candidate);
2927
- if (closed !== null) {
2928
- const parsed = tryParse(closed);
2929
- if (parsed !== UNPARSEABLE) return parsed;
2922
+ if (event.kind === "candidate-slot-closed") {
2923
+ const plannedSlot = plannedSlots.get(event.slotId);
2924
+ if (!plannedSlot) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} references unknown slot ${event.slotId}`);
2925
+ if (event.generationOperationId !== plannedSlot.generationOperationId) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} generation operation ${event.generationOperationId} does not match slot ${event.slotId} plan ${plannedSlot.generationOperationId}`);
2926
+ if (candidateBySlot.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was already bound to a candidate`);
2927
+ if (closedSlots.has(event.slotId)) throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} was closed twice`);
2928
+ const operation = operationsById.get(event.generationOperationId);
2929
+ if (!operation) throw new SearchLedgerIntegrityError(`candidate slot closure ${event.eventId} precedes operation ${event.generationOperationId}`);
2930
+ if (operation.outcome.status === "completed") throw new SearchLedgerIntegrityError(`candidate slot ${event.slotId} cannot close from completed operation ${event.generationOperationId}`);
2931
+ closedSlots.set(event.slotId, event);
2932
+ closedSlotEvents.push(event);
2933
+ return;
2930
2934
  }
2931
- const cut = lastCommaOutsideString(candidate);
2932
- if (cut <= 0) return UNPARSEABLE;
2933
- candidate = candidate.slice(0, cut);
2934
- }
2935
- return UNPARSEABLE;
2936
- }
2937
- //#endregion
2938
- //#region src/reflective-mutation.ts
2939
- /**
2940
- * Reflective mutation — primitives for trace-conditioned prompt rewriting.
2941
- *
2942
- * Used by `prompt-evolution.ts` (and any consumer running iterative
2943
- * improvement). Given a parent prompt + concrete trace evidence (top trials,
2944
- * bottom trials, missed expectations), produce an LLM-ready prompt that
2945
- * proposes targeted mutations — not blind rephrasings.
2946
- *
2947
- * Why this lives outside `prompt-evolution.ts`: any consumer that wants to
2948
- * run reflective rewriting WITHOUT the population/Pareto machinery can
2949
- * import these primitives directly.
2950
- *
2951
- * Quality bar (vs. naive "mutate this prompt"):
2952
- * - Show parent ↔ children diff, not just one variant
2953
- * - Quote specific missed goldens with their match phrases
2954
- * - Surface the model's actual emitted output side-by-side with what was expected
2955
- * - Quote concrete mutation primitives so the model has a vocabulary
2956
- */
2957
- /** Bound on rendered/carried `emitted` evidence. ONE constant shared by the
2958
- * producer (campaignBreakdown's per-scenario excerpt) and this renderer — if
2959
- * the two drifted, the tighter side would silently re-clip carried evidence. */
2960
- const EMITTED_EVIDENCE_MAX_CHARS = 2e3;
2961
- const DEFAULT_MUTATION_PRIMITIVES = [
2962
- "Strengthen an imperative (\"should\" → \"must\")",
2963
- "Add a concrete example pulled from a missed-golden phrase",
2964
- "Remove a redundant rule that did not improve recall",
2965
- "Add a counterfactual (\"if X is missing, the score is capped at Y\")",
2966
- "Reorder sections so the highest-impact rule is first",
2967
- "Replace abstract language with a domain-specific noun the trial misses"
2968
- ];
2969
- /**
2970
- * Build the LLM-ready reflection prompt. Output is plain text — pass it as
2971
- * the user message. The system message should be small and stable (e.g.
2972
- * "Output ONLY a JSON object matching the schema below.").
2973
- */
2974
- function buildReflectionPrompt(ctx) {
2975
- const primitives = ctx.mutationPrimitives ?? DEFAULT_MUTATION_PRIMITIVES;
2976
- const sections = [];
2977
- sections.push(`# Mutation target: ${ctx.target}`);
2978
- sections.push("");
2979
- sections.push(`You are tuning the prompt component named \`${ctx.target}\`. The current variant is shown below; you have ${ctx.topTrials.length} top trials and ${ctx.bottomTrials.length} bottom trials as evidence. Propose ${ctx.childCount} mutation${ctx.childCount === 1 ? "" : "s"} that fix specific weaknesses visible in the bottom trials. Avoid blank rephrasings.`);
2980
- sections.push("");
2981
- sections.push("## Current variant");
2982
- sections.push("```json");
2983
- sections.push(JSON.stringify(ctx.parentPayload, null, 2));
2984
- sections.push("```");
2985
- sections.push("");
2986
- if (ctx.bottomTrials.length > 0) {
2987
- sections.push("## Failures (bottom trials) — what went wrong");
2988
- sections.push("");
2989
- for (const trial of ctx.bottomTrials) {
2990
- sections.push(`### Trial \`${trial.id}\` — score ${trial.score.toFixed(2)}${trial.inputName ? ` (${trial.inputName})` : ""}`);
2991
- if (trial.failureNote) {
2992
- sections.push("");
2993
- sections.push(`**Why it scored low:** ${truncate(trial.failureNote, 1500)}`);
2994
- }
2995
- const missed = (trial.expectations ?? []).filter((e) => !e.matched);
2996
- if (missed.length > 0) {
2997
- sections.push("");
2998
- sections.push("**Missed expectations:**");
2999
- for (const m of missed) sections.push(`- \`${m.id}\`: should match phrase \`${quote(m.phrase)}\``);
2935
+ if (event.kind === "candidate-decided") {
2936
+ const candidate = candidates.get(event.candidateId);
2937
+ if (!candidate) throw new SearchLedgerIntegrityError(`decision ${event.eventId} references unknown candidate ${event.candidateId}`);
2938
+ if (candidate.decision) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} was decided twice`);
2939
+ if (event.decision.status === "selected") {
2940
+ if (!candidate.attempts.some((attempt) => attempt.outcome.status !== "errored")) throw new SearchLedgerIntegrityError(`candidate ${event.candidateId} cannot be selected without a measured task outcome`);
2941
+ if (decisions.some((decision) => decision.decision.status === "selected")) throw new SearchLedgerIntegrityError("more than one candidate was selected");
3000
2942
  }
3001
- if (trial.emitted) {
3002
- sections.push("");
3003
- sections.push("**What the agent emitted:**");
3004
- sections.push("```");
3005
- sections.push(truncate(trial.emitted, EMITTED_EVIDENCE_MAX_CHARS));
3006
- sections.push("```");
2943
+ candidate.decision = event;
2944
+ decisions.push(event);
2945
+ return;
2946
+ }
2947
+ const missingCandidateSlots = [...plannedSlots.values()].filter((slot) => !candidateBySlot.has(slot.slotId) && !closedSlots.has(slot.slotId)).map((slot) => slot.slotId);
2948
+ if (missingCandidateSlots.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing candidate slots: ${missingCandidateSlots.join(", ")}`);
2949
+ const missingTaskOutcomes = plannedTaskOutcomeKeys(planEvent, candidates);
2950
+ if (missingTaskOutcomes.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing task outcomes: ${missingTaskOutcomes.join(", ")}`);
2951
+ const missingOperations = [...plannedOperations.values()].filter((operation) => !operationsById.has(operation.operationId)).map((operation) => operation.operationId);
2952
+ if (missingOperations.length > 0) throw new SearchLedgerIntegrityError(`search completed with missing search operations: ${missingOperations.join(", ")}`);
2953
+ for (const operation of plannedOperations.values()) {
2954
+ if (operation.kind !== "candidate-generation") continue;
2955
+ const generationOutcome = operationsById.get(operation.operationId).outcome.status;
2956
+ const slots = [...plannedSlots.values()].filter((slot) => slot.generationOperationId === operation.operationId);
2957
+ if (slots.length === 0) continue;
2958
+ const registeredCount = slots.filter((slot) => candidateBySlot.has(slot.slotId)).length;
2959
+ const closedCount = slots.filter((slot) => closedSlots.has(slot.slotId)).length;
2960
+ if (generationOutcome === "completed" && closedCount > 0) throw new SearchLedgerIntegrityError(`completed generation operation ${operation.operationId} contains ${closedCount} closed slot(s)`);
2961
+ if (generationOutcome === "failed" && registeredCount > 0) throw new SearchLedgerIntegrityError(`failed generation operation ${operation.operationId} contains ${registeredCount} registered candidate(s)`);
2962
+ if (generationOutcome === "partial" && (registeredCount === 0 || closedCount === 0)) throw new SearchLedgerIntegrityError(`partial generation operation ${operation.operationId} must contain both a registered candidate and a closed slot`);
2963
+ }
2964
+ const pending = [...candidates.values()].filter((candidate) => candidate.decision === null);
2965
+ if (pending.length > 0) throw new SearchLedgerIntegrityError(`search completed with ${pending.length} candidate decision(s) missing`);
2966
+ const selected = decisions.filter((decision) => decision.decision.status === "selected");
2967
+ if (event.result.status === "selected") {
2968
+ if (selected.length !== 1 || selected[0].candidateId !== event.result.candidateId) throw new SearchLedgerIntegrityError(`search completion winner ${event.result.candidateId} does not match candidate decisions`);
2969
+ } else if (selected.length !== 0) throw new SearchLedgerIntegrityError("all-rejected completion contains a selected candidate");
2970
+ completion = event;
2971
+ };
2972
+ const finish = (entries) => {
2973
+ const selectedDecisions = decisions.filter((decision) => decision.decision.status === "selected");
2974
+ const rejectedDecisions = decisions.filter((decision) => decision.decision.status === "rejected");
2975
+ const outcomeCounts = {
2976
+ passed: 0,
2977
+ failed: 0,
2978
+ errored: 0
2979
+ };
2980
+ const operationOutcomeCounts = {
2981
+ completed: 0,
2982
+ partial: 0,
2983
+ failed: 0
2984
+ };
2985
+ let inputTokens = 0;
2986
+ let outputTokens = 0;
2987
+ let cachedTokens = 0;
2988
+ let costUsd = 0;
2989
+ const unknownTokenEventIds = [];
2990
+ const unknownCostEventIds = [];
2991
+ for (const attempt of attempts) outcomeCounts[attempt.outcome.status] += 1;
2992
+ for (const operation of operationEvents) operationOutcomeCounts[operation.outcome.status] += 1;
2993
+ for (const costedEvent of [...attempts, ...operationEvents]) {
2994
+ if (costedEvent.accounting.tokens.status === "known") {
2995
+ inputTokens += costedEvent.accounting.tokens.inputTokens;
2996
+ outputTokens += costedEvent.accounting.tokens.outputTokens;
2997
+ cachedTokens += costedEvent.accounting.tokens.cachedTokens;
2998
+ } else unknownTokenEventIds.push(costedEvent.eventId);
2999
+ if (costedEvent.accounting.cost.status === "known") costUsd += costedEvent.accounting.cost.usd;
3000
+ else {
3001
+ costUsd += costedEvent.accounting.cost.knownLowerBoundUsd;
3002
+ unknownCostEventIds.push(costedEvent.eventId);
3007
3003
  }
3008
- sections.push("");
3009
3004
  }
3010
- }
3011
- if (ctx.topTrials.length > 0) {
3012
- sections.push("## Successes (top trials) — what to preserve");
3013
- sections.push("");
3014
- for (const trial of ctx.topTrials) sections.push(`- \`${trial.id}\`: score ${trial.score.toFixed(2)}${trial.inputName ? ` (${trial.inputName})` : ""}`);
3015
- sections.push("");
3016
- }
3017
- sections.push("## Allowed mutation primitives");
3018
- sections.push("");
3019
- for (const p of primitives) sections.push(`- ${p}`);
3020
- sections.push("");
3021
- sections.push("## Output schema");
3022
- sections.push("");
3023
- sections.push("Respond with a JSON object — no prose, no markdown fences:");
3024
- sections.push("```json");
3025
- sections.push(JSON.stringify({ proposals: [{
3026
- label: "<short label, 40 chars>",
3027
- rationale: "<which failure this targets and which primitive you used>",
3028
- payload: "<full payload of the new variant same shape as the current variant>"
3029
- }] }, null, 2));
3030
- sections.push("```");
3031
- return sections.join("\n");
3032
- }
3033
- function truncate(s, max) {
3034
- if (s.length <= max) return s;
3035
- return `${s.slice(0, max)}… [truncated]`;
3036
- }
3037
- function quote(s) {
3038
- return s.replace(/`/g, "\\`");
3005
+ const accounting = unknownTokenEventIds.length === 0 && unknownCostEventIds.length === 0 ? {
3006
+ status: "known",
3007
+ inputTokens,
3008
+ outputTokens,
3009
+ cachedTokens,
3010
+ costUsd
3011
+ } : {
3012
+ status: "partial",
3013
+ knownInputTokens: inputTokens,
3014
+ knownOutputTokens: outputTokens,
3015
+ knownCachedTokens: cachedTokens,
3016
+ knownCostUsd: costUsd,
3017
+ unknownTokenEventIds,
3018
+ unknownCostEventIds
3019
+ };
3020
+ const selectedCandidateId = completion?.result.status === "selected" ? completion.result.candidateId : null;
3021
+ const status = completion?.result.status === "selected" ? "selected" : completion?.result.status === "all-rejected" ? "all-rejected" : "in-progress";
3022
+ const missingCandidateSlots = [...plannedSlots.values()].filter((slot) => !candidateBySlot.has(slot.slotId) && !closedSlots.has(slot.slotId)).map((slot) => slot.slotId);
3023
+ const missingTaskOutcomes = planEvent ? plannedTaskOutcomeKeys(planEvent, candidates) : [];
3024
+ const missingOperations = [...plannedOperations.values()].filter((operation) => !operationsById.has(operation.operationId)).map((operation) => operation.operationId);
3025
+ return {
3026
+ entries: [...entries],
3027
+ plan: planEvent,
3028
+ planExtensions,
3029
+ candidates: candidateEvents,
3030
+ closedCandidateSlots: closedSlotEvents,
3031
+ attempts,
3032
+ operations: operationEvents,
3033
+ decisions,
3034
+ completion,
3035
+ audit: {
3036
+ campaignId,
3037
+ eventCount: entries.length,
3038
+ candidateCount: candidates.size,
3039
+ closedCandidateSlotCount: closedSlots.size,
3040
+ attemptCount: attempts.length,
3041
+ operationCount: operationEvents.length,
3042
+ outcomes: outcomeCounts,
3043
+ operationOutcomes: operationOutcomeCounts,
3044
+ decisions: {
3045
+ selected: selectedDecisions.length,
3046
+ rejected: rejectedDecisions.length,
3047
+ pending: candidates.size - decisions.length
3048
+ },
3049
+ expected: {
3050
+ candidateSlots: plannedSlots.size,
3051
+ taskOutcomes: candidates.size * (planEvent?.plan.tasks.length ?? 0),
3052
+ operations: plannedOperations.size,
3053
+ missingCandidateSlots,
3054
+ missingTaskOutcomes,
3055
+ missingOperations
3056
+ },
3057
+ status,
3058
+ selectedCandidateId,
3059
+ accounting,
3060
+ headHash: entries.at(-1)?.entryHash ?? null
3061
+ }
3062
+ };
3063
+ };
3064
+ return {
3065
+ apply,
3066
+ finish
3067
+ };
3039
3068
  }
3040
- /**
3041
- * Parse the model's JSON response back into proposals. Tolerates markdown
3042
- * fences and surrounding prose. Returns at most `maxProposals`.
3043
- */
3044
- function parseReflectionResponse(raw, maxProposals) {
3045
- let text = raw.trim();
3046
- if (text.startsWith("```")) text = text.replace(/^```(?:json)?\n?/, "").replace(/\n?```$/, "");
3047
- let parsed = null;
3048
- const objectStart = text.indexOf("{");
3049
- const objectEnd = text.lastIndexOf("}");
3050
- const arrayStart = text.indexOf("[");
3051
- const arrayEnd = text.lastIndexOf("]");
3052
- const tryObjectFirst = objectStart >= 0 && (arrayStart < 0 || objectStart < arrayStart);
3053
- const candidates = [];
3054
- if (tryObjectFirst) {
3055
- if (objectStart >= 0 && objectEnd > objectStart) candidates.push(text.slice(objectStart, objectEnd + 1));
3056
- if (arrayStart >= 0 && arrayEnd > arrayStart) candidates.push(text.slice(arrayStart, arrayEnd + 1));
3057
- } else {
3058
- if (arrayStart >= 0 && arrayEnd > arrayStart) candidates.push(text.slice(arrayStart, arrayEnd + 1));
3059
- if (objectStart >= 0 && objectEnd > objectStart) candidates.push(text.slice(objectStart, objectEnd + 1));
3060
- }
3061
- for (const slice of candidates) try {
3062
- parsed = JSON.parse(slice);
3063
- break;
3064
- } catch {}
3065
- if (parsed == null) for (const slice of candidates) {
3066
- const closed = autoCloseTruncatedJson(slice);
3067
- if (closed != null && closed !== slice) try {
3068
- parsed = JSON.parse(closed);
3069
- break;
3070
- } catch {}
3071
- }
3072
- if (parsed == null) return [];
3073
- let proposalsRaw;
3074
- if (Array.isArray(parsed)) proposalsRaw = parsed;
3075
- else if (parsed && typeof parsed === "object") proposalsRaw = parsed.proposals;
3076
- if (!Array.isArray(proposalsRaw)) return [];
3077
- const out = [];
3078
- for (const p of proposalsRaw) {
3079
- if (!p || typeof p !== "object") continue;
3080
- const obj = p;
3081
- if (!("payload" in obj)) continue;
3082
- out.push({
3083
- label: typeof obj.label === "string" ? obj.label : "mutation",
3084
- rationale: typeof obj.rationale === "string" ? obj.rationale : "",
3085
- payload: obj.payload
3086
- });
3087
- if (maxProposals !== void 0 && out.length >= maxProposals) break;
3088
- }
3089
- return out;
3069
+ /** Every candidate slot must name a planned candidate-generation operation,
3070
+ * whether it arrives with the plan or with a later extension. */
3071
+ function assertSlotGenerationOperation(slot, plannedOperations) {
3072
+ if (plannedOperations.get(slot.generationOperationId)?.kind !== "candidate-generation") throw new SearchLedgerIntegrityError(`candidate slot ${slot.slotId} references unplanned candidate-generation operation ${slot.generationOperationId}`);
3090
3073
  }
3091
- //#endregion
3092
- //#region src/campaign/score-utils.ts
3093
- /**
3094
- * Shared campaign-score reductions used by every optimizer preset
3095
- * (`runOptimization`, external optimization methods, `compareOptimizationMethods`).
3096
- * "composite of a campaign" and "per-scenario / per-dimension breakdown" so
3097
- * the optimizers cannot drift on how a surface's score is computed.
3098
- */
3099
- /** Mean composite across cells with complete task-quality evidence.
3100
- * Partial judge results remain on their cells but never enter this value.
3101
- * A campaign with no complete score has no numeric mean and fails loudly. */
3102
- function campaignMeanComposite(campaign) {
3103
- const mean = campaignMeanCompositeOrNull(campaign);
3104
- if (mean === null) throw new Error("campaignMeanComposite: campaign has no complete cell-quality scores");
3105
- return mean;
3106
- }
3107
- /** Nullable campaign mean for wire and report fields that represent missing quality. */
3108
- function campaignMeanCompositeOrNull(campaign) {
3109
- const scores = campaign.cells.flatMap((cell) => {
3110
- const score = projectCampaignCellQuality(cell).score;
3111
- return score === void 0 ? [] : [score];
3112
- });
3113
- return scores.length === 0 ? null : scores.reduce((sum, score) => sum + score, 0) / scores.length;
3114
- }
3115
- /** Reject rank keys that cannot produce deterministic lexicographic ordering. */
3116
- function assertFiniteRankKey(key, label, expectedLength) {
3117
- if (!Array.isArray(key) || key.length === 0) throw new Error(`${label} must return a non-empty array`);
3118
- if (expectedLength !== void 0 && key.length !== expectedLength) throw new Error(`${label} returned ${key.length} elements; expected ${expectedLength}`);
3119
- for (let index = 0; index < key.length; index++) if (!Number.isFinite(key[index])) throw new Error(`${label}[${index}] must be finite`);
3120
- }
3121
- /** Compare fixed-length lexicographic rank keys where each element is higher-is-better.
3122
- * Returns a positive number when `a` ranks above `b`, negative when below, and
3123
- * zero when equal. */
3124
- function compareRankKeys(a, b) {
3125
- assertFiniteRankKey(a, "rank key a");
3126
- assertFiniteRankKey(b, "rank key b", a.length);
3127
- for (let i = 0; i < a.length; i++) {
3128
- const av = a[i];
3129
- const bv = b[i];
3130
- if (av !== bv) return av - bv;
3074
+ function plannedTaskOutcomeKeys(planEvent, candidates) {
3075
+ const missing = [];
3076
+ const registeredCandidates = [...candidates.values()].sort((a, b) => compareStrings(a.registered.slotId, b.registered.slotId));
3077
+ for (const candidate of registeredCandidates) {
3078
+ const slotId = candidate.registered.slotId;
3079
+ for (const task of planEvent.plan.tasks) if (!candidate.attempts.some((attempt) => attempt.task.taskId === task.taskId && attempt.outcome.status !== "errored")) missing.push(`${slotId}/${task.taskId}`);
3131
3080
  }
3132
- return 0;
3133
- }
3134
- /** Per-candidate evidence a reflective/patch proposer grounds its next proposal
3135
- * on: mean score per judge dimension + per-scenario composite. */
3136
- function campaignBreakdown(campaign) {
3137
- const dimSums = {};
3138
- const dimCounts = {};
3139
- const byScenario = /* @__PURE__ */ new Map();
3140
- const notesByScenario = /* @__PURE__ */ new Map();
3141
- const emittedByScenario = /* @__PURE__ */ new Map();
3142
- for (const cell of campaign.cells) {
3143
- const quality = projectCampaignCellQuality(cell);
3144
- if (quality.score === void 0) continue;
3145
- const judgeScores = Object.values(quality.successfulJudgeScores);
3146
- const cellComposite = quality.score;
3147
- const arr = byScenario.get(cell.scenarioId) ?? [];
3148
- arr.push(cellComposite);
3149
- byScenario.set(cell.scenarioId, arr);
3150
- if (typeof cell.artifact === "string" && cell.artifact.trim().length > 0) {
3151
- const prev = emittedByScenario.get(cell.scenarioId);
3152
- if (!prev || cellComposite < prev.composite) emittedByScenario.set(cell.scenarioId, {
3153
- composite: cellComposite,
3154
- text: cell.artifact.slice(0, EMITTED_EVIDENCE_MAX_CHARS)
3155
- });
3156
- }
3157
- for (const s of judgeScores) if (s.notes?.trim()) {
3158
- const set = notesByScenario.get(cell.scenarioId) ?? /* @__PURE__ */ new Set();
3159
- set.add(s.notes.trim());
3160
- notesByScenario.set(cell.scenarioId, set);
3081
+ return missing;
3082
+ }
3083
+ function normalizeEvent(event) {
3084
+ const artifacts = sortArtifacts(event.artifacts);
3085
+ if (event.kind === "search-planned") return {
3086
+ ...event,
3087
+ artifacts,
3088
+ plan: {
3089
+ candidateSlots: [...event.plan.candidateSlots].sort((a, b) => compareStrings(a.slotId, b.slotId)),
3090
+ tasks: [...event.plan.tasks].sort((a, b) => compareStrings(a.taskId, b.taskId)),
3091
+ operations: [...event.plan.operations].sort((a, b) => compareStrings(a.operationId, b.operationId))
3161
3092
  }
3162
- for (const score of judgeScores) for (const [key, value] of Object.entries(score.dimensions)) {
3163
- if (!Number.isFinite(value)) continue;
3164
- dimSums[key] = (dimSums[key] ?? 0) + value;
3165
- dimCounts[key] = (dimCounts[key] ?? 0) + 1;
3093
+ };
3094
+ if (event.kind === "search-plan-extended") return {
3095
+ ...event,
3096
+ artifacts,
3097
+ extension: {
3098
+ candidateSlots: [...event.extension.candidateSlots].sort((a, b) => compareStrings(a.slotId, b.slotId)),
3099
+ operations: [...event.extension.operations].sort((a, b) => compareStrings(a.operationId, b.operationId))
3166
3100
  }
3167
- }
3168
- const dimensions = {};
3169
- for (const key of Object.keys(dimSums)) {
3170
- const count = dimCounts[key] ?? 0;
3171
- dimensions[key] = count > 0 ? (dimSums[key] ?? 0) / count : 0;
3172
- }
3101
+ };
3102
+ if (event.kind === "candidate-registered") return {
3103
+ ...event,
3104
+ artifacts,
3105
+ lineage: {
3106
+ ...event.lineage,
3107
+ parentCandidateIds: sortedStrings(event.lineage.parentCandidateIds)
3108
+ },
3109
+ surfaces: [...event.surfaces].map((surface) => ({
3110
+ ...surface,
3111
+ artifact: { ...surface.artifact }
3112
+ })).sort((a, b) => compareStrings(a.surfaceId, b.surfaceId))
3113
+ };
3114
+ if (event.kind === "task-attempted") return {
3115
+ ...event,
3116
+ artifacts,
3117
+ surfaceEvidence: [...event.surfaceEvidence].map((evidence) => ({
3118
+ ...evidence,
3119
+ evidence: sortArtifacts(evidence.evidence)
3120
+ })).sort((a, b) => compareStrings(a.surfaceId, b.surfaceId))
3121
+ };
3173
3122
  return {
3174
- dimensions,
3175
- scenarios: [...byScenario.entries()].map(([scenarioId, comps]) => {
3176
- const notesSet = notesByScenario.get(scenarioId);
3177
- const notes = notesSet && notesSet.size > 0 ? [...notesSet].join(" | ") : void 0;
3178
- const emitted = emittedByScenario.get(scenarioId)?.text;
3179
- return {
3180
- scenarioId,
3181
- composite: comps.reduce((a, b) => a + b, 0) / comps.length,
3182
- ...notes ? { notes } : {},
3183
- ...emitted ? { emitted } : {}
3184
- };
3185
- })
3123
+ ...event,
3124
+ artifacts
3186
3125
  };
3187
3126
  }
3127
+ function sortArtifacts(artifacts) {
3128
+ return [...artifacts].map((artifact) => ({ ...artifact })).sort((a, b) => compareStrings(artifactKey(a), artifactKey(b)));
3129
+ }
3130
+ function artifactKey(artifact) {
3131
+ return canonicalString(artifact);
3132
+ }
3133
+ function compareStrings(a, b) {
3134
+ return a < b ? -1 : a > b ? 1 : 0;
3135
+ }
3136
+ function sortedStrings(values) {
3137
+ return [...values].sort();
3138
+ }
3139
+ function assertUnique(values, label, eventId) {
3140
+ if (new Set(values).size !== values.length) throw new SearchLedgerIntegrityError(`event ${eventId} contains duplicate ${label} values`);
3141
+ }
3142
+ function formatZodError(error) {
3143
+ return error.issues.map((issue) => `${issue.path.length > 0 ? issue.path.join(".") : "<root>"}: ${issue.message}`).join("; ");
3144
+ }
3188
3145
  //#endregion
3189
3146
  //#region src/campaign/search-history-receipt.ts
3190
3147
  const SEARCH_HISTORY_RECEIPT_SCHEMA_VERSION = "1.0.0";
3191
3148
  const SEARCH_HISTORY_RECEIPT_DIGEST_ALGORITHM = "rfc8785-sha256";
3149
+ function assertSearchHistoryAdmissionOptions(options) {
3150
+ if (options.searchHistoryPolicy !== void 0 && !["allow-missing", "require-complete"].includes(options.searchHistoryPolicy)) throw new Error(`unknown searchHistoryPolicy '${String(options.searchHistoryPolicy)}'`);
3151
+ if (options.searchHistoryVerification !== void 0 && !["receipt", "ledger"].includes(options.searchHistoryVerification)) throw new Error(`unknown searchHistoryVerification '${String(options.searchHistoryVerification)}'`);
3152
+ }
3153
+ /** Verify resolved bytes with the canonical codec, never a caller-supplied replay projection. */
3154
+ function verifySearchHistoryArtifact(receipt, storage) {
3155
+ verifySearchHistoryReceipt(receipt);
3156
+ const path = receipt.ledger.uri.startsWith("file:") ? fileURLToPath(receipt.ledger.uri) : receipt.ledger.uri;
3157
+ const text = storage.read(path);
3158
+ if (text === void 0) throw new Error(`search history ledger is missing or unreadable at '${path}'`);
3159
+ if (new TextEncoder().encode(text).byteLength !== receipt.ledger.byteLength) throw new Error("search history ledger byte length mismatch");
3160
+ if (hashCanonical(text) !== receipt.ledger.sha256) throw new Error("search history ledger digest mismatch");
3161
+ assertSearchHistoryMatchesReplay(receipt, replaySearchLedgerText(text, receipt.summary.campaignId, path));
3162
+ }
3192
3163
  var SearchHistoryRequiredError = class extends Error {
3193
3164
  producerId;
3194
3165
  reasons;
@@ -4696,666 +4667,23 @@ async function runImprovementLoop(opts) {
4696
4667
  };
4697
4668
  }
4698
4669
  //#endregion
4699
- //#region src/campaign/provenance.ts
4670
+ //#region src/campaign/gepa-candidate-population.ts
4700
4671
  /**
4701
- * Loop provenance the durable, queryable record of WHAT a self-improvement
4702
- * loop did and WHY, plus the OTel spans that let an OTLP collector pivot from
4703
- * an eval-run to the underlying candidate→cell→gate→promote chain.
4704
- *
4705
- * Two artifacts, one source of truth:
4706
- *
4707
- * 1. `LoopProvenanceRecord` — a structured JSON record capturing every
4708
- * candidate (surfaceHash + label + rationale + structured cause), its measured composite,
4709
- * the gate decision + reasons + delta, the held-out lift, the explicit
4710
- * baseline→candidate diff, and BACKEND PROVENANCE (the
4711
- * `assertRealBackend` verdict + worker call count + model). This is the
4712
- * ingestable audit artifact: the +lift recomputes from it, the "because
4713
- * Z" rationale survives in it, and a stub backend is detectable from it.
4714
- *
4715
- * 2. `loopProvenanceSpans()` — the same chain emitted as OTLP-ingestable
4716
- * `TraceSpanEvent`s, pivoted on the substrate's standard
4717
- * `tangle.runId` / `tangle.scenarioId` / `tangle.cellId` /
4718
- * `tangle.generation` attributes (the same pivots `/adapters/otel`
4719
- * reads). The hosted `/v1/ingest/traces` endpoint receives the FULL loop,
4720
- * not just the `cost.*` spans `runCampaign` already emits per cell.
4672
+ * Read GEPA's exact candidate graph from the artifact addressed by method provenance.
4721
4673
  *
4722
- * The record is built from the loop result and its settled cost receipts — no
4723
- * second usage collector can contradict what the measured cells recorded.
4674
+ * The reader checks the supplied digest, declared byte count, run identity,
4675
+ * candidate surfaces, parent graph, selection scores, and configured bounds.
4676
+ * This proves that the bytes match the supplied summary. The caller remains
4677
+ * responsible for obtaining that summary from trusted method provenance.
4724
4678
  */
4725
- /** One translation from a completed improvement loop into durable evidence. */
4726
- function loopProvenanceArgsFromResult(input) {
4727
- const { result } = input;
4728
- return {
4729
- runId: input.runId,
4730
- runDir: input.runDir,
4731
- timestamp: input.timestamp,
4732
- baselineSurface: input.baselineSurface,
4733
- winnerSurface: result.winnerSurface,
4734
- ...result.winnerLabel ? { winnerLabel: result.winnerLabel } : {},
4735
- ...result.winnerRationale ? { winnerRationale: result.winnerRationale } : {},
4736
- baselineSearchCampaign: result.baselineCampaign,
4737
- generations: result.generations.map(({ record, surfaces }) => ({
4738
- generationIndex: record.generationIndex,
4739
- candidates: record.candidates,
4740
- promoted: record.promoted,
4741
- surfaces: surfaces.map(({ surfaceHash, surface, campaign }) => ({
4742
- surfaceHash,
4743
- surface,
4744
- campaign
4745
- }))
4746
- })),
4747
- gate: result.gateResult,
4748
- ...result.holdout === "deferred" ? { holdout: "deferred" } : {},
4749
- baselineOnHoldout: result.baselineOnHoldout,
4750
- winnerOnHoldout: result.winnerOnHoldout,
4751
- ...result.neutralizedSurface && result.neutralizedOnHoldout ? {
4752
- neutralizedSurface: result.neutralizedSurface,
4753
- neutralizedOnHoldout: result.neutralizedOnHoldout
4754
- } : {},
4755
- costReceipts: input.costReceipts,
4756
- totalCostUsd: input.totalCostUsd,
4757
- totalDurationMs: input.totalDurationMs
4758
- };
4759
- }
4760
- function meanHoldoutComposite(campaign) {
4761
- return campaignMeanComposite(campaign);
4762
- }
4763
- /** Build the durable provenance record from a completed loop result. */
4764
- function buildLoopProvenanceRecord(args) {
4765
- if (!args.runId.trim() || !args.runDir.trim()) throw new Error("buildLoopProvenanceRecord: runId and runDir must be non-empty");
4766
- const timestampMs = Date.parse(args.timestamp);
4767
- if (!Number.isFinite(timestampMs) || new Date(timestampMs).toISOString() !== args.timestamp) throw new Error("buildLoopProvenanceRecord: timestamp must be a canonical ISO instant");
4768
- assertGateContributions(args.gate.contributingGates, "buildLoopProvenanceRecord");
4769
- const agentReceipts = args.costReceipts.filter((receipt) => receipt.channel === "agent");
4770
- const integrity = summarizeAgentReceiptIntegrity(agentReceipts);
4771
- const models = [...new Set(agentReceipts.map((receipt) => receipt.model))].sort();
4772
- const baselineSearchComposite = campaignMeanComposite(args.baselineSearchCampaign);
4773
- if (!Number.isFinite(baselineSearchComposite)) throw new Error("buildLoopProvenanceRecord: baselineSearchComposite must be finite");
4774
- const candidates = [];
4775
- let incumbentSurfaceHash = surfaceHash(args.baselineSurface);
4776
- let incumbentComposite = baselineSearchComposite;
4777
- let previousGeneration = -1;
4778
- for (const gen of args.generations) {
4779
- if (!Number.isSafeInteger(gen.generationIndex) || gen.generationIndex !== previousGeneration + 1) throw new Error("buildLoopProvenanceRecord: generation indices must be contiguous integers starting at zero");
4780
- previousGeneration = gen.generationIndex;
4781
- if (gen.candidates.length === 0) throw new Error("buildLoopProvenanceRecord: a recorded generation must contain a candidate");
4782
- if (new Set(gen.promoted).size !== gen.promoted.length || gen.promoted.length > 1) throw new Error("buildLoopProvenanceRecord: each generation may promote at most one candidate");
4783
- const promotedSet = new Set(gen.promoted);
4784
- const surfaceByHash = new Map(gen.surfaces.map((measured) => [measured.surfaceHash, measured]));
4785
- const candidateByHash = new Map(gen.candidates.map((candidate) => [candidate.surfaceHash, candidate]));
4786
- if (candidateByHash.size !== gen.candidates.length) throw new Error("buildLoopProvenanceRecord: duplicate candidate surface hash");
4787
- if (surfaceByHash.size !== gen.surfaces.length) throw new Error("buildLoopProvenanceRecord: duplicate candidate surface entry");
4788
- if (surfaceByHash.size !== candidateByHash.size) throw new Error("buildLoopProvenanceRecord: every measured candidate requires exactly one surface");
4789
- for (const promotedHash of promotedSet) if (!candidateByHash.has(promotedHash)) throw new Error("buildLoopProvenanceRecord: promoted hash has no measured candidate");
4790
- for (const c of gen.candidates) {
4791
- validateCandidateMeasurement(c, incumbentSurfaceHash, incumbentComposite, promotedSet.has(c.surfaceHash));
4792
- const measured = surfaceByHash.get(c.surfaceHash);
4793
- if (measured === void 0) throw new Error("buildLoopProvenanceRecord: measured candidate is missing its surface");
4794
- const { surface, campaign } = measured;
4795
- if (!surfaceHashMatches(surface, c.surfaceHash)) throw new Error("buildLoopProvenanceRecord: candidate surface hash does not match its surface bytes");
4796
- if (campaign.splitDigest !== args.baselineSearchCampaign.splitDigest) throw new Error("buildLoopProvenanceRecord: candidate campaign does not match the search split");
4797
- const entry = {
4798
- generation: gen.generationIndex,
4799
- surfaceHash: c.surfaceHash,
4800
- contentHash: surfaceContentHash(surface),
4801
- campaignDigest: campaignMeasurementDigest(campaign),
4802
- parentSurfaceHash: c.parentSurfaceHash,
4803
- parentComposite: c.parentComposite,
4804
- eligibleForPromotion: c.eligibleForPromotion,
4805
- coverage: {
4806
- expectedCells: c.coverage.expectedCells,
4807
- scorableCells: c.coverage.scorableCells,
4808
- unscorableCells: c.coverage.unscorableCells.map((cell) => ({ ...cell }))
4809
- },
4810
- composite: c.composite,
4811
- promoted: promotedSet.has(c.surfaceHash)
4812
- };
4813
- if (c.label) entry.label = c.label;
4814
- if (c.rationale) entry.rationale = c.rationale;
4815
- if (c.attribution) entry.attribution = c.attribution;
4816
- if (c.observedDeltaFromParent !== void 0) entry.observedDeltaFromParent = c.observedDeltaFromParent;
4817
- candidates.push(entry);
4818
- }
4819
- const promotedHash = gen.promoted[0];
4820
- if (promotedHash) {
4821
- const promoted = candidateByHash.get(promotedHash);
4822
- incumbentSurfaceHash = promoted.surfaceHash;
4823
- if (promoted.composite === null) throw new Error("buildLoopProvenanceRecord: promoted candidate is missing a composite");
4824
- incumbentComposite = promoted.composite;
4825
- }
4826
- }
4827
- if (surfaceHash(args.winnerSurface) !== incumbentSurfaceHash) throw new Error("buildLoopProvenanceRecord: winner surface does not match the final promoted incumbent");
4828
- const holdoutDeferred = args.holdout === "deferred";
4829
- if (args.baselineOnHoldout.splitDigest !== args.winnerOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: baseline and winner use different holdout splits");
4830
- if (args.neutralizedSurface === void 0 !== (args.neutralizedOnHoldout === void 0)) throw new Error("buildLoopProvenanceRecord: neutralized surface and campaign must be supplied together");
4831
- if (args.neutralizedOnHoldout && args.neutralizedOnHoldout.splitDigest !== args.baselineOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: neutralized campaign uses a different holdout split");
4832
- if (holdoutDeferred && args.neutralizedOnHoldout) throw new Error("buildLoopProvenanceRecord: a deferred holdout cannot include a neutralized measurement");
4833
- const holdoutMeasurement = holdoutDeferred ? { kind: "deferred" } : {
4834
- kind: "measured",
4835
- baseline: meanHoldoutComposite(args.baselineOnHoldout),
4836
- winner: meanHoldoutComposite(args.winnerOnHoldout),
4837
- ...args.neutralizedOnHoldout ? { neutralized: meanHoldoutComposite(args.neutralizedOnHoldout) } : {}
4838
- };
4839
- const diff = surfaceContentHash(args.baselineSurface) === surfaceContentHash(args.winnerSurface) ? "" : renderSurfaceDiff(args.winnerSurface, args.baselineSurface);
4840
- const recordWithoutDigest = {
4841
- schema: "tangle.loop-provenance",
4842
- runId: args.runId,
4843
- runDir: args.runDir,
4844
- timestamp: args.timestamp,
4845
- baselineContentHash: surfaceContentHash(args.baselineSurface),
4846
- winnerContentHash: surfaceContentHash(args.winnerSurface),
4847
- diff,
4848
- candidates,
4849
- evidence: {
4850
- search: {
4851
- splitDigest: args.baselineSearchCampaign.splitDigest,
4852
- baselineCampaignDigest: campaignMeasurementDigest(args.baselineSearchCampaign)
4853
- },
4854
- holdout: {
4855
- splitDigest: args.baselineOnHoldout.splitDigest,
4856
- baselineCampaignDigest: campaignMeasurementDigest(args.baselineOnHoldout),
4857
- winnerCampaignDigest: campaignMeasurementDigest(args.winnerOnHoldout),
4858
- ...args.neutralizedSurface && args.neutralizedOnHoldout && holdoutMeasurement.kind === "measured" && holdoutMeasurement.neutralized !== void 0 ? { neutralized: {
4859
- contentHash: surfaceContentHash(args.neutralizedSurface),
4860
- campaignDigest: campaignMeasurementDigest(args.neutralizedOnHoldout),
4861
- composite: holdoutMeasurement.neutralized,
4862
- lift: holdoutMeasurement.neutralized - holdoutMeasurement.baseline
4863
- } } : {}
4864
- },
4865
- costReceiptsDigest: canonicalDigest([...args.costReceipts].sort((left, right) => compareCodeUnits(left.callId, right.callId)))
4866
- },
4867
- baselineSearchComposite,
4868
- gate: {
4869
- decision: args.gate.decision,
4870
- reasons: args.gate.reasons,
4871
- ...args.gate.delta === void 0 ? {} : { delta: args.gate.delta },
4872
- contributingGates: args.gate.contributingGates.map((g) => ({
4873
- name: g.name,
4874
- status: g.status,
4875
- detail: durableGateDetail(g.detail)
4876
- }))
4877
- },
4878
- ...holdoutMeasurement.kind === "deferred" ? { holdout: "deferred" } : {
4879
- baselineHoldoutComposite: holdoutMeasurement.baseline,
4880
- winnerHoldoutComposite: holdoutMeasurement.winner,
4881
- heldOutLift: holdoutMeasurement.winner - holdoutMeasurement.baseline
4882
- },
4883
- backend: {
4884
- verdict: integrity.verdict,
4885
- workerCallCount: integrity.totalRecords,
4886
- models,
4887
- totalInputTokens: integrity.totalInputTokens,
4888
- totalOutputTokens: integrity.totalOutputTokens,
4889
- totalCostUsd: integrity.totalCostUsd
4890
- },
4891
- totalCostUsd: args.totalCostUsd,
4892
- totalDurationMs: args.totalDurationMs
4893
- };
4894
- if (args.optimizationMethod) recordWithoutDigest.optimizationMethod = durableOptimizationMethod(args.optimizationMethod);
4895
- if (args.winnerLabel) recordWithoutDigest.winnerLabel = args.winnerLabel;
4896
- if (args.winnerRationale) recordWithoutDigest.winnerRationale = args.winnerRationale;
4897
- return {
4898
- ...recordWithoutDigest,
4899
- recordDigest: canonicalDigest(recordWithoutDigest)
4900
- };
4901
- }
4902
- function durableOptimizationMethod(value) {
4903
- if (!value || typeof value !== "object" || typeof value.name !== "string" || !value.name.trim() || value.name.trim() !== value.name) throw new Error("buildLoopProvenanceRecord: optimization method name is invalid");
4904
- if (!value.cost || !Number.isFinite(value.cost.totalCostUsd) || value.cost.totalCostUsd < 0 || typeof value.cost.accountingComplete !== "boolean" || !Array.isArray(value.cost.incompleteReasons) || value.cost.incompleteReasons.some((reason) => typeof reason !== "string" || !reason.trim()) || value.cost.accountingComplete !== (value.cost.incompleteReasons.length === 0)) throw new Error("buildLoopProvenanceRecord: optimization method cost is invalid");
4905
- if (value.durationMs !== void 0 && (!Number.isFinite(value.durationMs) || value.durationMs < 0)) throw new Error("buildLoopProvenanceRecord: optimization method duration is invalid");
4906
- try {
4907
- return JSON.parse(canonicalString(value));
4908
- } catch (cause) {
4909
- throw new Error("buildLoopProvenanceRecord: optimization method data must be canonical JSON", { cause });
4910
- }
4911
- }
4912
- /** Digest the exact campaign fields that can affect a measured comparison. */
4913
- function campaignMeasurementDigest(campaign) {
4914
- assertCampaignSplitIdentity(campaign.scenarios, campaign.reps, campaign.splitDigest);
4915
- return canonicalDigest({
4916
- schema: "tangle.campaign-measurement",
4917
- manifestHash: campaign.manifestHash,
4918
- splitDigest: campaign.splitDigest,
4919
- seed: campaign.seed,
4920
- reps: campaign.reps,
4921
- runDir: campaign.runDir,
4922
- scenarios: campaign.scenarios,
4923
- cells: [...campaign.cells].sort((left, right) => compareCodeUnits(left.cellId, right.cellId)).map((cell) => ({
4924
- manifestHash: cell.manifestHash ?? null,
4925
- cellId: cell.cellId,
4926
- scenarioId: cell.scenarioId,
4927
- rep: cell.rep,
4928
- generation: cell.generation ?? null,
4929
- judgeScores: cell.judgeScores,
4930
- costUsd: cell.costUsd,
4931
- costProvenance: cell.costProvenance,
4932
- costCallIds: [...cell.costCallIds ?? []].sort(),
4933
- tokenUsage: cell.tokenUsage,
4934
- resolvedModels: [...cell.resolvedModels ?? []].sort(),
4935
- resolvedModel: cell.resolvedModel ?? null,
4936
- durationMs: cell.durationMs,
4937
- seed: cell.seed,
4938
- cached: cell.cached,
4939
- errorStage: cell.errorStage ?? null,
4940
- errorJudge: cell.errorJudge ?? null,
4941
- error: cell.error ?? null
4942
- }))
4943
- });
4944
- }
4945
- /** Recompute and validate the self-addressed durable record. */
4946
- function verifyLoopProvenanceRecord(record) {
4947
- if (record.schema !== "tangle.loop-provenance") throw new Error("loop provenance has an unsupported schema");
4948
- const { recordDigest, ...recordWithoutDigest } = record;
4949
- if (recordDigest !== canonicalDigest(recordWithoutDigest)) throw new Error("loop provenance record digest does not match its contents");
4950
- assertGateContributions(record.gate?.contributingGates, "loop provenance");
4951
- return record;
4952
- }
4953
- /** SHA-256 over the RFC 8785 canonical JSON of `value`. Throws
4954
- * `LedgerCanonicalizationError` for a value with no canonical form. */
4955
- function canonicalDigest(value) {
4956
- return hashCanonical(value);
4957
- }
4958
- function durableGateDetail(detail) {
4959
- if (detail === void 0) return null;
4960
- try {
4961
- return JSON.parse(canonicalString(detail));
4962
- } catch (cause) {
4963
- throw new Error("buildLoopProvenanceRecord: gate detail must be canonical JSON", { cause });
4964
- }
4965
- }
4966
- function assertGateContributions(value, source) {
4967
- if (!Array.isArray(value)) throw new Error(`${source}: gate contributingGates must be an array`);
4968
- const statuses = /* @__PURE__ */ new Set([
4969
- "pass",
4970
- "fail",
4971
- "not_evaluated"
4972
- ]);
4973
- for (const [index, contribution] of value.entries()) {
4974
- if (!contribution || typeof contribution !== "object") throw new Error(`${source}: gate contribution ${index} must be an object`);
4975
- const item = contribution;
4976
- if (typeof item.name !== "string" || item.name.length === 0) throw new Error(`${source}: gate contribution ${index} must have a non-empty name`);
4977
- if (!statuses.has(String(item.status))) throw new Error(`${source}: gate contribution '${item.name}' must have status pass, fail, or not_evaluated`);
4978
- if ("passed" in item) throw new Error(`${source}: gate contribution '${item.name}' uses obsolete passed; use status instead`);
4979
- }
4980
- }
4981
- function validateCandidateMeasurement(candidate, expectedParentHash, expectedParentComposite, promoted) {
4982
- if (!candidate.parentSurfaceHash || !/^[a-f0-9]{16}$/.test(candidate.parentSurfaceHash)) throw new Error("buildLoopProvenanceRecord: parentSurfaceHash must be 16 lowercase hex characters");
4983
- if (candidate.parentSurfaceHash !== expectedParentHash) throw new Error("buildLoopProvenanceRecord: candidate parent does not match the incumbent");
4984
- if (candidate.parentComposite === void 0 || !Number.isFinite(candidate.parentComposite) || Math.abs(candidate.parentComposite - expectedParentComposite) > 1e-12) throw new Error("buildLoopProvenanceRecord: candidate parentComposite does not match the incumbent");
4985
- if (candidate.observedDeltaFromParent !== void 0) {
4986
- if (!Number.isFinite(candidate.observedDeltaFromParent)) throw new Error("buildLoopProvenanceRecord: observedDeltaFromParent must be finite");
4987
- if (candidate.eligibleForPromotion !== true) throw new Error("buildLoopProvenanceRecord: observedDeltaFromParent requires a complete eligible candidate and parentSurfaceHash");
4988
- }
4989
- const coverage = candidate.coverage;
4990
- if (!Number.isSafeInteger(coverage.expectedCells) || coverage.expectedCells <= 0 || !Number.isSafeInteger(coverage.scorableCells) || coverage.scorableCells < 0 || coverage.scorableCells > coverage.expectedCells) throw new Error("buildLoopProvenanceRecord: invalid candidate coverage denominator");
4991
- const unscorableIds = /* @__PURE__ */ new Set();
4992
- for (const failure of coverage.unscorableCells) {
4993
- if (typeof failure.cellId !== "string" || failure.cellId.length === 0 || typeof failure.reason !== "string" || failure.reason.length === 0 || unscorableIds.has(failure.cellId)) throw new Error("buildLoopProvenanceRecord: invalid candidate coverage failures");
4994
- unscorableIds.add(failure.cellId);
4995
- }
4996
- if (coverage.expectedCells - coverage.scorableCells !== coverage.unscorableCells.length) throw new Error("buildLoopProvenanceRecord: candidate coverage counts do not match its failures");
4997
- const complete = coverage.scorableCells === coverage.expectedCells && coverage.unscorableCells.length === 0;
4998
- if (candidate.eligibleForPromotion !== complete) throw new Error("buildLoopProvenanceRecord: candidate eligibility contradicts its coverage receipt");
4999
- if (complete) {
5000
- if (candidate.composite === null || !Number.isFinite(candidate.composite)) throw new Error("buildLoopProvenanceRecord: complete candidate composite must be finite");
5001
- if (candidate.observedDeltaFromParent === void 0) throw new Error("buildLoopProvenanceRecord: complete candidate is missing observedDeltaFromParent");
5002
- const recomputed = candidate.composite - candidate.parentComposite;
5003
- if (Math.abs(candidate.observedDeltaFromParent - recomputed) > 1e-12) throw new Error("buildLoopProvenanceRecord: observed delta does not match measured scores");
5004
- } else {
5005
- if (candidate.composite !== null && !Number.isFinite(candidate.composite)) throw new Error("buildLoopProvenanceRecord: candidate composite must be finite or null");
5006
- if (candidate.observedDeltaFromParent !== void 0) throw new Error("buildLoopProvenanceRecord: incomplete candidate cannot carry observed delta");
5007
- }
5008
- if (promoted && (!complete || (candidate.observedDeltaFromParent ?? 0) <= 0)) throw new Error("buildLoopProvenanceRecord: promoted candidate must improve the incumbent");
5009
- }
5010
- function hashId(parts) {
5011
- return createHash("sha256").update(parts.join(":")).digest("hex");
5012
- }
5013
- /**
5014
- * Build the loop's OTLP-ingestable spans from a provenance record. One root
5015
- * span per loop (`tangle.runId`), one span per generation, one span per
5016
- * candidate (carrying its surfaceHash + label), and one span for the gate
5017
- * decision (carrying reasons + delta + lift). Candidate + gate spans pivot on
5018
- * the same `tangle.runId` / `tangle.generation` attributes `/adapters/otel`
5019
- * reads, so the hosted collector reconstructs the full tree.
5020
- *
5021
- * Times are synthesized monotonically off a single base so the span tree is
5022
- * orderable; the substrate does not retain per-candidate wall-clock starts.
5023
- */
5024
- function loopProvenanceSpans(record, opts = {}) {
5025
- const traceId = hashId(["trace", record.runId]).slice(0, 32);
5026
- const baseTimeMs = opts.baseTimeMs ?? (Date.parse(record.timestamp) || Date.now());
5027
- const durationMs = Math.max(1, record.totalDurationMs);
5028
- if (!Number.isSafeInteger(baseTimeMs) || baseTimeMs < 0) throw new RangeError("loop provenance baseTimeMs must be a non-negative safe integer");
5029
- if (!Number.isSafeInteger(durationMs)) throw new RangeError("loop provenance duration must be a safe integer number of milliseconds");
5030
- const baseTime = BigInt(baseTimeMs);
5031
- const baseNano = (baseTime * 1000000n).toString();
5032
- const endNano = ((baseTime + BigInt(durationMs)) * 1000000n).toString();
5033
- const spans = [];
5034
- const rootSpanId = hashId(["root", record.runId]).slice(0, 16);
5035
- const rootAttributes = {
5036
- "tangle.runId": record.runId,
5037
- "tangle.runDir": record.runDir,
5038
- "tangle.baselineContentHash": record.baselineContentHash,
5039
- "tangle.winnerContentHash": record.winnerContentHash,
5040
- "tangle.baselineSearchComposite": record.baselineSearchComposite,
5041
- "tangle.gateDecision": record.gate.decision,
5042
- "tangle.backendVerdict": record.backend.verdict,
5043
- "tangle.workerCallCount": record.backend.workerCallCount,
5044
- "tangle.totalCostUsd": record.totalCostUsd
5045
- };
5046
- if (record.heldOutLift !== void 0) rootAttributes["tangle.heldOutLift"] = record.heldOutLift;
5047
- if (record.holdout) rootAttributes["tangle.holdout"] = record.holdout;
5048
- spans.push({
5049
- traceId,
5050
- spanId: rootSpanId,
5051
- name: "improvement-loop",
5052
- startTimeUnixNano: baseNano,
5053
- endTimeUnixNano: endNano,
5054
- attributes: rootAttributes,
5055
- status: { code: "OK" },
5056
- "tangle.runId": record.runId
5057
- });
5058
- const byGen = /* @__PURE__ */ new Map();
5059
- for (const c of record.candidates) {
5060
- const arr = byGen.get(c.generation) ?? [];
5061
- arr.push(c);
5062
- byGen.set(c.generation, arr);
5063
- }
5064
- for (const [generation, cands] of [...byGen.entries()].sort((a, b) => a[0] - b[0])) {
5065
- const genSpanId = hashId([
5066
- "gen",
5067
- record.runId,
5068
- String(generation)
5069
- ]).slice(0, 16);
5070
- const measuredComposites = cands.flatMap((candidate) => candidate.composite === null ? [] : [candidate.composite]);
5071
- spans.push({
5072
- traceId,
5073
- spanId: genSpanId,
5074
- parentSpanId: rootSpanId,
5075
- name: `generation-${generation}`,
5076
- startTimeUnixNano: baseNano,
5077
- endTimeUnixNano: endNano,
5078
- attributes: {
5079
- "tangle.runId": record.runId,
5080
- "tangle.generation": generation,
5081
- "tangle.populationSize": cands.length,
5082
- ...measuredComposites.length > 0 ? { "tangle.bestComposite": Math.max(...measuredComposites) } : {}
5083
- },
5084
- "tangle.runId": record.runId,
5085
- "tangle.generation": generation
5086
- });
5087
- for (let i = 0; i < cands.length; i++) {
5088
- const c = cands[i];
5089
- const candSpanId = hashId([
5090
- "cand",
5091
- record.runId,
5092
- String(generation),
5093
- c.surfaceHash
5094
- ]).slice(0, 16);
5095
- const attributes = {
5096
- "tangle.runId": record.runId,
5097
- "tangle.generation": generation,
5098
- "tangle.surfaceHash": c.surfaceHash,
5099
- "tangle.contentHash": c.contentHash,
5100
- "tangle.parentSurfaceHash": c.parentSurfaceHash,
5101
- "tangle.parentComposite": c.parentComposite,
5102
- "tangle.eligibleForPromotion": c.eligibleForPromotion,
5103
- "tangle.expectedCells": c.coverage.expectedCells,
5104
- "tangle.scorableCells": c.coverage.scorableCells,
5105
- "tangle.unscorableCells": c.coverage.unscorableCells.length,
5106
- "tangle.promoted": c.promoted
5107
- };
5108
- if (c.composite !== null) attributes["tangle.composite"] = c.composite;
5109
- if (c.observedDeltaFromParent !== void 0) attributes["tangle.observedDeltaFromParent"] = c.observedDeltaFromParent;
5110
- if (c.label) attributes["tangle.candidateLabel"] = c.label;
5111
- if (c.rationale) attributes["tangle.candidateRationale"] = c.rationale;
5112
- spans.push({
5113
- traceId,
5114
- spanId: candSpanId,
5115
- parentSpanId: genSpanId,
5116
- name: `candidate-${c.surfaceHash}`,
5117
- startTimeUnixNano: baseNano,
5118
- endTimeUnixNano: endNano,
5119
- attributes,
5120
- "tangle.runId": record.runId,
5121
- "tangle.generation": generation
5122
- });
5123
- }
5124
- }
5125
- const gateSpanId = hashId(["gate", record.runId]).slice(0, 16);
5126
- const gateAttributes = {
5127
- "tangle.runId": record.runId,
5128
- "tangle.gateDecision": record.gate.decision,
5129
- "tangle.gateReasons": JSON.stringify(record.gate.reasons)
5130
- };
5131
- const gateDelta = record.gate.delta ?? record.heldOutLift;
5132
- if (gateDelta !== void 0) gateAttributes["tangle.gateDelta"] = gateDelta;
5133
- if (record.heldOutLift !== void 0) gateAttributes["tangle.heldOutLift"] = record.heldOutLift;
5134
- if (record.baselineHoldoutComposite !== void 0) gateAttributes["tangle.baselineHoldoutComposite"] = record.baselineHoldoutComposite;
5135
- if (record.winnerHoldoutComposite !== void 0) gateAttributes["tangle.winnerHoldoutComposite"] = record.winnerHoldoutComposite;
5136
- if (record.holdout) gateAttributes["tangle.holdout"] = record.holdout;
5137
- spans.push({
5138
- traceId,
5139
- spanId: gateSpanId,
5140
- parentSpanId: rootSpanId,
5141
- name: "gate-decision",
5142
- startTimeUnixNano: endNano,
5143
- endTimeUnixNano: endNano,
5144
- attributes: gateAttributes,
5145
- status: { code: "OK" },
5146
- "tangle.runId": record.runId
5147
- });
5148
- return spans;
5149
- }
5150
- /** Canonical durable paths under the run dir. */
5151
- function provenanceRecordPath(runDir) {
5152
- return join(runDir, "loop-provenance.json");
5153
- }
5154
- /**
5155
- * Canonical path for the durable OTLP spans JSONL file under a loop run directory.
5156
- */
5157
- function provenanceSpansPath(runDir) {
5158
- return join(runDir, "loop-provenance-spans.jsonl");
5159
- }
5160
- /** Snapshot a held-out campaign into the hosted `EvalRunGenerationSnapshot`
5161
- * shape — per-cell composite + per-judge dimensions, aggregate mean, cost,
5162
- * duration. The dashboard renders these as the baseline → winner comparison. */
5163
- function snapshotFromHoldout(index, surfaceHash, surface, campaign) {
5164
- return {
5165
- index,
5166
- surfaceHash,
5167
- surface,
5168
- cells: campaign.cells.map((cell) => {
5169
- const execution = campaignCellExecutionEvidence(cell);
5170
- const quality = projectCampaignCellQuality(cell);
5171
- const score = {
5172
- scenarioId: cell.scenarioId,
5173
- rep: cell.rep,
5174
- compositeMean: quality.score ?? null,
5175
- dimensions: quality.judgeScores?.perJudge ?? {},
5176
- terminalOutcome: execution.terminalOutcome,
5177
- executionErrorCount: execution.executionErrorCount ?? null
5178
- };
5179
- if (cell.error) score.errorMessage = cell.error;
5180
- return score;
5181
- }),
5182
- compositeMean: campaignMeanCompositeOrNull(campaign),
5183
- costUsd: campaign.aggregates.cost.totalCostUsd,
5184
- durationMs: campaign.durationMs
5185
- };
5186
- }
5187
- /** Build the hosted `EvalRunEvent` from the loop args + record — baseline +
5188
- * winner snapshots, gate decision, held-out lift, cost, duration. Shipped to
5189
- * `/v1/ingest/eval-runs` so the run appears in the dashboard's run list (the
5190
- * trace spans, shipped separately, back the per-candidate drill-down). */
5191
- function buildEvalRunEvent(args, record) {
5192
- return {
5193
- runId: args.runId,
5194
- runDir: args.runDir,
5195
- timestamp: args.timestamp,
5196
- status: "finished",
5197
- labels: {},
5198
- baseline: snapshotFromHoldout(0, record.baselineContentHash, args.baselineSurface, args.baselineOnHoldout),
5199
- generations: [snapshotFromHoldout(1, record.winnerContentHash, args.winnerSurface, args.winnerOnHoldout)],
5200
- gateDecision: args.gate.decision,
5201
- ...record.heldOutLift !== void 0 ? { holdoutLift: record.heldOutLift } : {},
5202
- totalCostUsd: args.totalCostUsd,
5203
- totalDurationMs: args.totalDurationMs
5204
- };
5205
- }
5206
- /**
5207
- * Build the provenance record + OTel spans and persist them durably under the
5208
- * run dir (and ship spans to a hosted collector when one is wired). Returns
5209
- * both artifacts so the caller can assert on / re-derive from them.
5210
- *
5211
- * Fail-loud: the durable write throws on storage failure (a swallowed write is
5212
- * exactly the "emitted but lost" failure this closes). The hosted span ship is
5213
- * the one best-effort leg — its failure is logged, not thrown, so an offline
5214
- * collector never fails the loop (the durable artifact is the source of truth).
5215
- */
5216
- async function emitLoopProvenance(args) {
5217
- const record = buildLoopProvenanceRecord(args);
5218
- const spans = loopProvenanceSpans(record);
5219
- args.storage.ensureDir(args.runDir);
5220
- const recordPath = provenanceRecordPath(args.runDir);
5221
- const spansPath = provenanceSpansPath(args.runDir);
5222
- args.storage.write(recordPath, JSON.stringify(record, null, 2));
5223
- args.storage.write(spansPath, spans.map((s) => JSON.stringify(s)).join("\n"));
5224
- if (args.hostedClient) {
5225
- try {
5226
- await args.hostedClient.ingestEvalRun(buildEvalRunEvent(args, record));
5227
- } catch (err) {
5228
- const msg = err instanceof Error ? err.message : String(err);
5229
- console.warn(`[agent-eval] hosted eval-run ingest failed (continuing): ${msg}`);
5230
- }
5231
- try {
5232
- await args.hostedClient.ingestTraces(spans);
5233
- } catch (err) {
5234
- const msg = err instanceof Error ? err.message : String(err);
5235
- console.warn(`[agent-eval] provenance span ingest failed (continuing): ${msg}`);
5236
- }
5237
- }
5238
- return {
5239
- record,
5240
- spans,
5241
- recordPath,
5242
- spansPath
5243
- };
5244
- }
5245
- //#endregion
5246
- //#region src/campaign/optimization-cost.ts
5247
- /** Attribute method calls while retaining the shared account's admission and read behavior. */
5248
- function createMethodCostScope(account, methodName) {
5249
- const tags = { optimizationAttempt: crypto.randomUUID() };
5250
- return {
5251
- ledger: Object.freeze({
5252
- costCeilingUsd: account.costCeilingUsd,
5253
- runPaidCall: (input) => account.runPaidCall({
5254
- ...input,
5255
- tags: {
5256
- ...input.tags,
5257
- ...tags
5258
- }
5259
- }),
5260
- summary: account.summary.bind(account),
5261
- list: account.list.bind(account),
5262
- reconcile: account.reconcile.bind(account),
5263
- markCompleted: account.markCompleted.bind(account),
5264
- costPerCompletedTask: account.costPerCompletedTask.bind(account),
5265
- ...account.listPending ? { listPending: account.listPending.bind(account) } : {},
5266
- ...account.waitForIdle ? { waitForIdle: account.waitForIdle.bind(account) } : {}
5267
- }),
5268
- reconcile(reported) {
5269
- const summary = account.summary({ tags });
5270
- if (summary.pendingCalls > 0) throw new Error(`optimization method '${methodName}' returned with ${summary.pendingCalls} pending paid call(s)`);
5271
- const recorded = costFromLedgerSummary(summary);
5272
- const totalCostUsd = Math.max(reported.totalCostUsd, recorded.totalCostUsd);
5273
- const roundingToleranceUsd = Number.EPSILON * Math.max(1, totalCostUsd) * (summary.totalCalls + 1);
5274
- const combined = combineComparisonCosts([{
5275
- label: "reported",
5276
- cost: reported
5277
- }, {
5278
- label: "recorded",
5279
- cost: recorded
5280
- }]);
5281
- const incompleteReasons = [
5282
- ...reported.incompleteReasons,
5283
- ...recorded.incompleteReasons.map((reason) => `recorded: ${reason}`),
5284
- ...recorded.totalCostUsd - reported.totalCostUsd > roundingToleranceUsd ? [`reported ${reported.totalCostUsd} USD below recorded ${recorded.totalCostUsd} USD`] : []
5285
- ];
5286
- return {
5287
- totalCostUsd,
5288
- costProvenance: combined.costProvenance.kind === "uncaptured" ? combined.costProvenance : {
5289
- kind: combined.costProvenance.kind,
5290
- usd: totalCostUsd
5291
- },
5292
- accountingComplete: incompleteReasons.length === 0,
5293
- incompleteReasons
5294
- };
5295
- }
5296
- };
5297
- }
5298
- /** Keep the cost fields a custom optimization method must report. */
5299
- function costFromLedgerSummary(summary) {
5300
- const cost = {
5301
- totalCostUsd: summary.totalCostUsd,
5302
- costProvenance: structuredClone(summary.costProvenance),
5303
- accountingComplete: summary.accountingComplete,
5304
- incompleteReasons: [...summary.incompleteReasons]
5305
- };
5306
- assertComparisonCost(cost, "cost ledger");
5307
- return cost;
5308
- }
5309
- /** Combine method costs without turning one unknown bill into a known total. */
5310
- function combineComparisonCosts(entries) {
5311
- const totalCostUsd = entries.reduce((total, entry) => total + entry.cost.totalCostUsd, 0);
5312
- const cost = {
5313
- totalCostUsd,
5314
- costProvenance: entries.some((entry) => entry.cost.costProvenance.kind === "uncaptured") ? {
5315
- kind: "uncaptured",
5316
- usd: null
5317
- } : entries.every((entry) => entry.cost.costProvenance.kind === "observed") ? {
5318
- kind: "observed",
5319
- usd: totalCostUsd
5320
- } : {
5321
- kind: "estimated",
5322
- usd: totalCostUsd
5323
- },
5324
- accountingComplete: entries.every((entry) => entry.cost.accountingComplete),
5325
- incompleteReasons: entries.flatMap((entry) => entry.cost.incompleteReasons.map((reason) => `${entry.label}: ${reason}`))
5326
- };
5327
- assertComparisonCost(cost, "combined cost");
5328
- return cost;
5329
- }
5330
- function assertComparisonCost(cost, label) {
5331
- if (!cost || typeof cost !== "object") throw new Error(`compareOptimizationMethods: ${label} returned no cost`);
5332
- if (!Number.isFinite(cost.totalCostUsd) || cost.totalCostUsd < 0) throw new Error(`compareOptimizationMethods: ${label} returned an invalid totalCostUsd`);
5333
- const provenance = cost.costProvenance;
5334
- if (!provenance || typeof provenance !== "object" || provenance.kind !== "observed" && provenance.kind !== "estimated" && provenance.kind !== "uncaptured" || (provenance.kind === "uncaptured" ? provenance.usd !== null : !Number.isFinite(provenance.usd) || provenance.usd < 0)) throw new Error(`compareOptimizationMethods: ${label} returned invalid costProvenance`);
5335
- if (provenance.kind !== "uncaptured" && provenance.usd !== cost.totalCostUsd) throw new Error(`compareOptimizationMethods: ${label} returned costProvenance inconsistent with totalCostUsd`);
5336
- if (typeof cost.accountingComplete !== "boolean") throw new Error(`compareOptimizationMethods: ${label} returned invalid accountingComplete`);
5337
- if (!Array.isArray(cost.incompleteReasons) || cost.incompleteReasons.some((reason) => typeof reason !== "string" || reason.trim().length === 0)) throw new Error(`compareOptimizationMethods: ${label} returned invalid incompleteReasons`);
5338
- if (cost.accountingComplete !== (cost.incompleteReasons.length === 0)) throw new Error(`compareOptimizationMethods: ${label} returned inconsistent cost completeness and reasons`);
5339
- if (cost.accountingComplete && provenance.kind === "uncaptured") throw new Error(`compareOptimizationMethods: ${label} cannot mark uncaptured cost as complete`);
5340
- }
5341
- //#endregion
5342
- //#region src/campaign/gepa-candidate-population.ts
5343
- /**
5344
- * Read GEPA's exact candidate graph from the artifact addressed by method provenance.
5345
- *
5346
- * The reader checks the supplied digest, declared byte count, run identity,
5347
- * candidate surfaces, parent graph, selection scores, and configured bounds.
5348
- * This proves that the bytes match the supplied summary. The caller remains
5349
- * responsible for obtaining that summary from trusted method provenance.
5350
- */
5351
- function readGepaCandidatePopulationArtifact(input) {
5352
- assertGepaCandidatePopulationSummary(input.summary);
5353
- const scenarioIds = scenarioIdSet(input.summary.scenarioIds);
5354
- const storage = input.storage ?? fsCampaignStorage();
5355
- const contents = storage.read(input.summary.path);
5356
- if (contents === void 0) {
5357
- const state = storage.exists(input.summary.path) ? "unreadable" : "missing";
5358
- throw new Error(`GEPA candidate population artifact is ${state} at '${input.summary.path}'`);
4679
+ function readGepaCandidatePopulationArtifact(input) {
4680
+ assertGepaCandidatePopulationSummary(input.summary);
4681
+ const scenarioIds = scenarioIdSet(input.summary.scenarioIds);
4682
+ const storage = input.storage ?? fsCampaignStorage();
4683
+ const contents = storage.read(input.summary.path);
4684
+ if (contents === void 0) {
4685
+ const state = storage.exists(input.summary.path) ? "unreadable" : "missing";
4686
+ throw new Error(`GEPA candidate population artifact is ${state} at '${input.summary.path}'`);
5359
4687
  }
5360
4688
  const bytes = new TextEncoder().encode(contents).byteLength;
5361
4689
  if (bytes !== input.summary.bytes) throw new Error(`GEPA candidate population byte count mismatch at '${input.summary.path}': expected ${input.summary.bytes}, got ${bytes}`);
@@ -5524,6 +4852,213 @@ function assertPositiveSafeInteger(value, label) {
5524
4852
  if (!Number.isSafeInteger(value) || value <= 0) throw new Error(`GEPA candidate population ${label} must be a positive safe integer`);
5525
4853
  }
5526
4854
  //#endregion
4855
+ //#region src/campaign/optimization-cost.ts
4856
+ /** Attribute method calls while retaining the shared account's admission and read behavior. */
4857
+ function createMethodCostScope(account, methodName) {
4858
+ const tags = { optimizationAttempt: crypto.randomUUID() };
4859
+ return {
4860
+ ledger: Object.freeze({
4861
+ costCeilingUsd: account.costCeilingUsd,
4862
+ runPaidCall: (input) => account.runPaidCall({
4863
+ ...input,
4864
+ tags: {
4865
+ ...input.tags,
4866
+ ...tags
4867
+ }
4868
+ }),
4869
+ summary: account.summary.bind(account),
4870
+ list: account.list.bind(account),
4871
+ reconcile: account.reconcile.bind(account),
4872
+ markCompleted: account.markCompleted.bind(account),
4873
+ costPerCompletedTask: account.costPerCompletedTask.bind(account),
4874
+ ...account.listPending ? { listPending: account.listPending.bind(account) } : {},
4875
+ ...account.waitForIdle ? { waitForIdle: account.waitForIdle.bind(account) } : {}
4876
+ }),
4877
+ reconcile(reported) {
4878
+ const summary = account.summary({ tags });
4879
+ if (summary.pendingCalls > 0) throw new Error(`optimization method '${methodName}' returned with ${summary.pendingCalls} pending paid call(s)`);
4880
+ const recorded = costFromLedgerSummary(summary);
4881
+ const totalCostUsd = Math.max(reported.totalCostUsd, recorded.totalCostUsd);
4882
+ const roundingToleranceUsd = Number.EPSILON * Math.max(1, totalCostUsd) * (summary.totalCalls + 1);
4883
+ const combined = combineComparisonCosts([{
4884
+ label: "reported",
4885
+ cost: reported
4886
+ }, {
4887
+ label: "recorded",
4888
+ cost: recorded
4889
+ }]);
4890
+ const incompleteReasons = [
4891
+ ...reported.incompleteReasons,
4892
+ ...recorded.incompleteReasons.map((reason) => `recorded: ${reason}`),
4893
+ ...recorded.totalCostUsd - reported.totalCostUsd > roundingToleranceUsd ? [`reported ${reported.totalCostUsd} USD below recorded ${recorded.totalCostUsd} USD`] : []
4894
+ ];
4895
+ return {
4896
+ totalCostUsd,
4897
+ costProvenance: combined.costProvenance.kind === "uncaptured" ? combined.costProvenance : {
4898
+ kind: combined.costProvenance.kind,
4899
+ usd: totalCostUsd
4900
+ },
4901
+ accountingComplete: incompleteReasons.length === 0,
4902
+ incompleteReasons
4903
+ };
4904
+ }
4905
+ };
4906
+ }
4907
+ /** Keep the cost fields a custom optimization method must report. */
4908
+ function costFromLedgerSummary(summary) {
4909
+ const cost = {
4910
+ totalCostUsd: summary.totalCostUsd,
4911
+ costProvenance: structuredClone(summary.costProvenance),
4912
+ accountingComplete: summary.accountingComplete,
4913
+ incompleteReasons: [...summary.incompleteReasons]
4914
+ };
4915
+ assertComparisonCost(cost, "cost ledger");
4916
+ return cost;
4917
+ }
4918
+ /** Combine method costs without turning one unknown bill into a known total. */
4919
+ function combineComparisonCosts(entries) {
4920
+ const totalCostUsd = entries.reduce((total, entry) => total + entry.cost.totalCostUsd, 0);
4921
+ const cost = {
4922
+ totalCostUsd,
4923
+ costProvenance: entries.some((entry) => entry.cost.costProvenance.kind === "uncaptured") ? {
4924
+ kind: "uncaptured",
4925
+ usd: null
4926
+ } : entries.every((entry) => entry.cost.costProvenance.kind === "observed") ? {
4927
+ kind: "observed",
4928
+ usd: totalCostUsd
4929
+ } : {
4930
+ kind: "estimated",
4931
+ usd: totalCostUsd
4932
+ },
4933
+ accountingComplete: entries.every((entry) => entry.cost.accountingComplete),
4934
+ incompleteReasons: entries.flatMap((entry) => entry.cost.incompleteReasons.map((reason) => `${entry.label}: ${reason}`))
4935
+ };
4936
+ assertComparisonCost(cost, "combined cost");
4937
+ return cost;
4938
+ }
4939
+ function assertComparisonCost(cost, label) {
4940
+ if (!cost || typeof cost !== "object") throw new Error(`compareOptimizationMethods: ${label} returned no cost`);
4941
+ if (!Number.isFinite(cost.totalCostUsd) || cost.totalCostUsd < 0) throw new Error(`compareOptimizationMethods: ${label} returned an invalid totalCostUsd`);
4942
+ const provenance = cost.costProvenance;
4943
+ if (!provenance || typeof provenance !== "object" || provenance.kind !== "observed" && provenance.kind !== "estimated" && provenance.kind !== "uncaptured" || (provenance.kind === "uncaptured" ? provenance.usd !== null : !Number.isFinite(provenance.usd) || provenance.usd < 0)) throw new Error(`compareOptimizationMethods: ${label} returned invalid costProvenance`);
4944
+ if (provenance.kind !== "uncaptured" && provenance.usd !== cost.totalCostUsd) throw new Error(`compareOptimizationMethods: ${label} returned costProvenance inconsistent with totalCostUsd`);
4945
+ if (typeof cost.accountingComplete !== "boolean") throw new Error(`compareOptimizationMethods: ${label} returned invalid accountingComplete`);
4946
+ if (!Array.isArray(cost.incompleteReasons) || cost.incompleteReasons.some((reason) => typeof reason !== "string" || reason.trim().length === 0)) throw new Error(`compareOptimizationMethods: ${label} returned invalid incompleteReasons`);
4947
+ if (cost.accountingComplete !== (cost.incompleteReasons.length === 0)) throw new Error(`compareOptimizationMethods: ${label} returned inconsistent cost completeness and reasons`);
4948
+ if (cost.accountingComplete && provenance.kind === "uncaptured") throw new Error(`compareOptimizationMethods: ${label} cannot mark uncaptured cost as complete`);
4949
+ }
4950
+ //#endregion
4951
+ //#region src/campaign/optimization-method.ts
4952
+ /** Both complete-method workflows use the same detached inputs, accounting, and evidence admission. */
4953
+ async function executeOptimizationMethod(options) {
4954
+ assertSearchHistoryAdmissionOptions(options);
4955
+ const { method, input } = options;
4956
+ const costScope = createMethodCostScope(input.costLedger, method.name);
4957
+ const cloneScenarios = (scenarios) => Object.freeze(scenarios.map((scenario) => structuredClone(scenario)));
4958
+ const selected = structuredClone(await method.optimize(Object.freeze({
4959
+ ...input,
4960
+ baselineSurface: structuredClone(input.baselineSurface),
4961
+ trainScenarios: cloneScenarios(input.trainScenarios),
4962
+ selectionScenarios: cloneScenarios(input.selectionScenarios),
4963
+ judges: Object.freeze(input.judges.map((judge) => {
4964
+ const dimensions = judge.dimensions.map((dimension) => Object.freeze({ ...dimension }));
4965
+ Object.freeze(dimensions);
4966
+ return Object.freeze({
4967
+ ...judge,
4968
+ dimensions
4969
+ });
4970
+ })),
4971
+ runOptions: Object.freeze({ ...input.runOptions }),
4972
+ costLedger: costScope.ledger
4973
+ })));
4974
+ assertOptimizationResult(method.name, selected);
4975
+ const history = searchHistoryCoverageRow(method.name, selected.searchHistory);
4976
+ if (options.searchHistoryPolicy === "require-complete") assertCompleteSearchHistory(method.name, selected.searchHistory);
4977
+ if (options.searchHistoryVerification === "ledger") {
4978
+ if (!selected.searchHistory) assertCompleteSearchHistory(method.name, selected.searchHistory);
4979
+ verifySearchHistoryArtifact(selected.searchHistory, options.storage);
4980
+ }
4981
+ return {
4982
+ selected,
4983
+ cost: costScope.reconcile(selected.cost),
4984
+ history: options.searchHistoryVerification === "ledger" ? Object.freeze({
4985
+ ...history,
4986
+ ledgerVerified: true
4987
+ }) : history
4988
+ };
4989
+ }
4990
+ function assertOptimizationResult(name, result) {
4991
+ if (!result || typeof result !== "object") throw new Error(`compareOptimizationMethods: method '${name}' returned no result`);
4992
+ try {
4993
+ surfaceContentHash(result.winnerSurface);
4994
+ } catch (cause) {
4995
+ throw new Error(`compareOptimizationMethods: method '${name}' returned an invalid winnerSurface`, { cause });
4996
+ }
4997
+ assertComparisonCost(result.cost, `method '${name}'`);
4998
+ if (result.durationMs !== void 0 && (!Number.isFinite(result.durationMs) || result.durationMs < 0)) throw new Error(`compareOptimizationMethods: method '${name}' returned an invalid durationMs`);
4999
+ if (result.provenance !== void 0) assertOptimizationProvenance(name, result.provenance);
5000
+ }
5001
+ function assertOptimizationProvenance(methodName, value) {
5002
+ const fail = (field) => {
5003
+ throw new Error(`compareOptimizationMethods: method '${methodName}' returned invalid provenance.${field}`);
5004
+ };
5005
+ if (!value || typeof value !== "object") fail("value");
5006
+ if (value.source?.kind !== "package" || !["observed", "declared"].includes(value.source.evidence) || typeof value.source.package !== "string" || !value.source.package.trim() || typeof value.source.version !== "string" || !value.source.version.trim()) fail("source");
5007
+ for (const [field, entry] of [["sourceUrl", value.source.sourceUrl], ["revision", value.source.revision]]) if (entry !== void 0 && (typeof entry !== "string" || !entry.trim())) fail(`source.${field}`);
5008
+ if (typeof value.runId !== "string" || !value.runId.trim()) fail("runId");
5009
+ if (value.optimizerModel !== void 0 && (typeof value.optimizerModel !== "string" || !value.optimizerModel.trim() || value.optimizerModel.trim() !== value.optimizerModel)) fail("optimizerModel");
5010
+ if (value.optimizerCallRef !== void 0 && (typeof value.optimizerCallRef !== "string" || !value.optimizerCallRef.trim() || value.optimizerCallRef.trim() !== value.optimizerCallRef)) fail("optimizerCallRef");
5011
+ if (typeof value.resumed !== "boolean") fail("resumed");
5012
+ if (value.seedApplied !== void 0 && typeof value.seedApplied !== "boolean") fail("seedApplied");
5013
+ if (!Number.isSafeInteger(value.evaluationCount) || value.evaluationCount < 0) fail("evaluationCount");
5014
+ if (typeof value.artifactDir !== "string" || !value.artifactDir.trim()) fail("artifactDir");
5015
+ if (value.tokenUsage !== void 0) {
5016
+ for (const field of [
5017
+ "inputTokens",
5018
+ "outputTokens",
5019
+ "totalTokens",
5020
+ "calls"
5021
+ ]) if (!Number.isSafeInteger(value.tokenUsage[field]) || value.tokenUsage[field] < 0) fail(`tokenUsage.${field}`);
5022
+ for (const field of [
5023
+ "cachedInputTokens",
5024
+ "cacheWriteInputTokens",
5025
+ "reasoningTokens"
5026
+ ]) {
5027
+ const entry = value.tokenUsage[field];
5028
+ if (entry !== void 0 && (!Number.isSafeInteger(entry) || entry < 0)) fail(`tokenUsage.${field}`);
5029
+ }
5030
+ if ((value.tokenUsage.cachedInputTokens ?? 0) + (value.tokenUsage.cacheWriteInputTokens ?? 0) > value.tokenUsage.inputTokens) fail("tokenUsage.inputTokens");
5031
+ if (value.tokenUsage.reasoningTokens !== void 0 && value.tokenUsage.reasoningTokens > value.tokenUsage.outputTokens) fail("tokenUsage.reasoningTokens");
5032
+ if (value.tokenUsage.totalTokens !== value.tokenUsage.inputTokens + value.tokenUsage.outputTokens) fail("tokenUsage.totalTokens");
5033
+ }
5034
+ if (value.observations !== void 0) {
5035
+ if (value.observations.scope !== "callback-submitted-candidates" || typeof value.observations.path !== "string" || !value.observations.path.trim() || typeof value.observations.sha256 !== "string" || !/^sha256:[0-9a-f]{64}$/.test(value.observations.sha256)) fail("observations");
5036
+ for (const field of [
5037
+ "submittedCandidates",
5038
+ "evaluations",
5039
+ "refusals"
5040
+ ]) if (!Number.isSafeInteger(value.observations[field]) || value.observations[field] < 0) fail(`observations.${field}`);
5041
+ }
5042
+ if (value.gepaCandidatePopulation !== void 0) {
5043
+ try {
5044
+ assertGepaCandidatePopulationSummary(value.gepaCandidatePopulation);
5045
+ } catch {
5046
+ fail("gepaCandidatePopulation");
5047
+ }
5048
+ if (value.gepaCandidatePopulation.runId !== value.runId) fail("gepaCandidatePopulation.runId");
5049
+ }
5050
+ if (value.modelExecutions !== void 0) {
5051
+ if (value.modelExecutions.scope !== "runtime-model-calls" || typeof value.modelExecutions.path !== "string" || !value.modelExecutions.path.trim() || typeof value.modelExecutions.sha256 !== "string" || !/^sha256:[0-9a-f]{64}$/.test(value.modelExecutions.sha256)) fail("modelExecutions");
5052
+ for (const field of [
5053
+ "calls",
5054
+ "succeeded",
5055
+ "failed"
5056
+ ]) if (!Number.isSafeInteger(value.modelExecutions[field]) || value.modelExecutions[field] < 0) fail(`modelExecutions.${field}`);
5057
+ if (value.modelExecutions.calls !== value.modelExecutions.succeeded + value.modelExecutions.failed) fail("modelExecutions.calls");
5058
+ }
5059
+ if (value.optimizerModel === void 0 !== (value.optimizerCallRef === void 0) || value.optimizerModel === void 0 !== (value.modelExecutions === void 0)) fail("optimizerModel execution provenance");
5060
+ }
5061
+ //#endregion
5527
5062
  //#region src/campaign/presets/compare-optimization-methods.ts
5528
5063
  /**
5529
5064
  * Compare optimization methods on shared train, selection, and test data.
@@ -5538,7 +5073,7 @@ async function compareOptimizationMethods(opts) {
5538
5073
  assertOptimizationMethods(opts.methods);
5539
5074
  assertComparisonPartitions(opts);
5540
5075
  const searchHistoryPolicy = opts.searchHistoryPolicy ?? "allow-missing";
5541
- if (searchHistoryPolicy !== "allow-missing" && searchHistoryPolicy !== "require-complete") throw new TypeError(`compareOptimizationMethods: unknown searchHistoryPolicy '${String(searchHistoryPolicy)}'`);
5076
+ assertSearchHistoryAdmissionOptions(opts);
5542
5077
  const seed = opts.seed ?? 42;
5543
5078
  const confidence = opts.confidence ?? .95;
5544
5079
  assertConfidence(confidence);
@@ -5548,6 +5083,8 @@ async function compareOptimizationMethods(opts) {
5548
5083
  const minimumResamples = minimumBootstrapResamples(confidence, comparisonCount);
5549
5084
  const resamples = opts.resamples ?? Math.max(2e3, minimumResamples);
5550
5085
  assertComparisonControls(opts, seed, resamples, confidence);
5086
+ const evidence = opts.evidence === void 0 ? void 0 : structuredClone(opts.evidence);
5087
+ const evidenceBySurface = /* @__PURE__ */ new Map();
5551
5088
  const storage = opts.storage ?? fsCampaignStorage();
5552
5089
  const resolvedRunDir = resolveRunDir(opts.runDir, opts.repo);
5553
5090
  const baselineSurface = structuredClone(opts.baselineSurface);
@@ -5569,6 +5106,11 @@ async function compareOptimizationMethods(opts) {
5569
5106
  runDir: `${resolvedRunDir}/${tag}`
5570
5107
  });
5571
5108
  assertCompleteCampaign(campaign, opts.testScenarios, opts.reps ?? 1, true, `compareOptimizationMethods: ${tag} final comparison`);
5109
+ if (evidence) evidenceBySurface.set(surfaceContentHash(measuredSurface), createCampaignEvidenceReceipt({
5110
+ campaign,
5111
+ surface: measuredSurface,
5112
+ context: evidence
5113
+ }));
5572
5114
  const byScenario = {};
5573
5115
  for (const { scenarioId, composite } of campaignBreakdown(campaign).scenarios) byScenario[scenarioId] = composite;
5574
5116
  return byScenario;
@@ -5582,16 +5124,18 @@ async function compareOptimizationMethods(opts) {
5582
5124
  const optimizationOwner = new AbortController();
5583
5125
  const optimized = await mapConcurrent(opts.methods, optimizationConcurrency, async (method) => {
5584
5126
  try {
5585
- const methodCost = createMethodCostScope(costLedger, method.name);
5586
- const out = await method.optimize(createOptimizationMethodInput(opts, method.name, resolvedRunDir, seed, baselineSurface, methodCost.ledger, optimizationOwner.signal));
5587
- assertOptimizationResult(method.name, out);
5588
- const searchHistoryCoverage = searchHistoryCoverageRow(method.name, out.searchHistory);
5589
- if (searchHistoryPolicy === "require-complete") assertCompleteSearchHistory(method.name, out.searchHistory);
5590
- const winnerSurface = structuredClone(out.winnerSurface);
5127
+ const { selected: out, cost, history: searchHistoryCoverage } = await executeOptimizationMethod({
5128
+ method,
5129
+ input: createOptimizationMethodInput(opts, method.name, resolvedRunDir, seed, baselineSurface, costLedger, optimizationOwner.signal),
5130
+ storage,
5131
+ searchHistoryPolicy: opts.searchHistoryPolicy,
5132
+ searchHistoryVerification: opts.searchHistoryVerification
5133
+ });
5134
+ const winnerSurface = out.winnerSurface;
5591
5135
  return {
5592
5136
  name: method.name,
5593
5137
  winnerSurface,
5594
- cost: methodCost.reconcile(out.cost),
5138
+ cost,
5595
5139
  ...out.durationMs === void 0 ? {} : { durationMs: out.durationMs },
5596
5140
  ...out.provenance === void 0 ? {} : { provenance: out.provenance },
5597
5141
  searchHistoryCoverage
@@ -5647,6 +5191,15 @@ async function compareOptimizationMethods(opts) {
5647
5191
  winnerSurface: structuredClone(w.winnerSurface),
5648
5192
  rank: 0
5649
5193
  };
5194
+ if (evidence) {
5195
+ const baseline = evidenceBySurface.get(surfaceContentHash(baselineSurface));
5196
+ const winner = evidenceBySurface.get(surfaceContentHash(w.winnerSurface));
5197
+ if (!baseline || !winner) throw new Error("final measurement evidence is missing");
5198
+ score.evidence = {
5199
+ baseline,
5200
+ winner
5201
+ };
5202
+ }
5650
5203
  if (w.durationMs !== void 0) score.durationMs = w.durationMs;
5651
5204
  if (w.provenance !== void 0) score.provenance = structuredClone(w.provenance);
5652
5205
  return score;
@@ -5739,77 +5292,6 @@ function assertOptimizationMethods(methods) {
5739
5292
  pathOwners.set(pathKey, method.name);
5740
5293
  }
5741
5294
  }
5742
- function assertOptimizationResult(name, result) {
5743
- if (!result || typeof result !== "object") throw new Error(`compareOptimizationMethods: method '${name}' returned no result`);
5744
- try {
5745
- surfaceContentHash(result.winnerSurface);
5746
- } catch (cause) {
5747
- throw new Error(`compareOptimizationMethods: method '${name}' returned an invalid winnerSurface`, { cause });
5748
- }
5749
- assertComparisonCost(result.cost, `method '${name}'`);
5750
- if (result.durationMs !== void 0 && (!Number.isFinite(result.durationMs) || result.durationMs < 0)) throw new Error(`compareOptimizationMethods: method '${name}' returned an invalid durationMs`);
5751
- if (result.provenance !== void 0) assertOptimizationProvenance(name, result.provenance);
5752
- }
5753
- function assertOptimizationProvenance(methodName, value) {
5754
- const fail = (field) => {
5755
- throw new Error(`compareOptimizationMethods: method '${methodName}' returned invalid provenance.${field}`);
5756
- };
5757
- if (!value || typeof value !== "object") fail("value");
5758
- if (value.source?.kind !== "package" || !["observed", "declared"].includes(value.source.evidence) || typeof value.source.package !== "string" || !value.source.package.trim() || typeof value.source.version !== "string" || !value.source.version.trim()) fail("source");
5759
- for (const [field, entry] of [["sourceUrl", value.source.sourceUrl], ["revision", value.source.revision]]) if (entry !== void 0 && (typeof entry !== "string" || !entry.trim())) fail(`source.${field}`);
5760
- if (typeof value.runId !== "string" || !value.runId.trim()) fail("runId");
5761
- if (value.optimizerModel !== void 0 && (typeof value.optimizerModel !== "string" || !value.optimizerModel.trim() || value.optimizerModel.trim() !== value.optimizerModel)) fail("optimizerModel");
5762
- if (value.optimizerCallRef !== void 0 && (typeof value.optimizerCallRef !== "string" || !value.optimizerCallRef.trim() || value.optimizerCallRef.trim() !== value.optimizerCallRef)) fail("optimizerCallRef");
5763
- if (typeof value.resumed !== "boolean") fail("resumed");
5764
- if (value.seedApplied !== void 0 && typeof value.seedApplied !== "boolean") fail("seedApplied");
5765
- if (!Number.isSafeInteger(value.evaluationCount) || value.evaluationCount < 0) fail("evaluationCount");
5766
- if (typeof value.artifactDir !== "string" || !value.artifactDir.trim()) fail("artifactDir");
5767
- if (value.tokenUsage !== void 0) {
5768
- for (const field of [
5769
- "inputTokens",
5770
- "outputTokens",
5771
- "totalTokens",
5772
- "calls"
5773
- ]) if (!Number.isSafeInteger(value.tokenUsage[field]) || value.tokenUsage[field] < 0) fail(`tokenUsage.${field}`);
5774
- for (const field of [
5775
- "cachedInputTokens",
5776
- "cacheWriteInputTokens",
5777
- "reasoningTokens"
5778
- ]) {
5779
- const entry = value.tokenUsage[field];
5780
- if (entry !== void 0 && (!Number.isSafeInteger(entry) || entry < 0)) fail(`tokenUsage.${field}`);
5781
- }
5782
- if ((value.tokenUsage.cachedInputTokens ?? 0) + (value.tokenUsage.cacheWriteInputTokens ?? 0) > value.tokenUsage.inputTokens) fail("tokenUsage.inputTokens");
5783
- if (value.tokenUsage.reasoningTokens !== void 0 && value.tokenUsage.reasoningTokens > value.tokenUsage.outputTokens) fail("tokenUsage.reasoningTokens");
5784
- if (value.tokenUsage.totalTokens !== value.tokenUsage.inputTokens + value.tokenUsage.outputTokens) fail("tokenUsage.totalTokens");
5785
- }
5786
- if (value.observations !== void 0) {
5787
- if (value.observations.scope !== "callback-submitted-candidates" || typeof value.observations.path !== "string" || !value.observations.path.trim() || typeof value.observations.sha256 !== "string" || !/^sha256:[0-9a-f]{64}$/.test(value.observations.sha256)) fail("observations");
5788
- for (const field of [
5789
- "submittedCandidates",
5790
- "evaluations",
5791
- "refusals"
5792
- ]) if (!Number.isSafeInteger(value.observations[field]) || value.observations[field] < 0) fail(`observations.${field}`);
5793
- }
5794
- if (value.gepaCandidatePopulation !== void 0) {
5795
- try {
5796
- assertGepaCandidatePopulationSummary(value.gepaCandidatePopulation);
5797
- } catch {
5798
- fail("gepaCandidatePopulation");
5799
- }
5800
- if (value.gepaCandidatePopulation.runId !== value.runId) fail("gepaCandidatePopulation.runId");
5801
- }
5802
- if (value.modelExecutions !== void 0) {
5803
- if (value.modelExecutions.scope !== "runtime-model-calls" || typeof value.modelExecutions.path !== "string" || !value.modelExecutions.path.trim() || typeof value.modelExecutions.sha256 !== "string" || !/^sha256:[0-9a-f]{64}$/.test(value.modelExecutions.sha256)) fail("modelExecutions");
5804
- for (const field of [
5805
- "calls",
5806
- "succeeded",
5807
- "failed"
5808
- ]) if (!Number.isSafeInteger(value.modelExecutions[field]) || value.modelExecutions[field] < 0) fail(`modelExecutions.${field}`);
5809
- if (value.modelExecutions.calls !== value.modelExecutions.succeeded + value.modelExecutions.failed) fail("modelExecutions.calls");
5810
- }
5811
- if (value.optimizerModel === void 0 !== (value.optimizerCallRef === void 0) || value.optimizerModel === void 0 !== (value.modelExecutions === void 0)) fail("optimizerModel execution provenance");
5812
- }
5813
5295
  function assertComparisonControls(opts, seed, resamples, confidence) {
5814
5296
  if (opts.optimizationRunOptions && "costCeiling" in opts.optimizationRunOptions) throw new Error("compareOptimizationMethods: optimizationRunOptions.costCeiling is not supported; costCeiling covers optimization and final scoring");
5815
5297
  if (!opts.judges || opts.judges.length === 0) throw new Error("compareOptimizationMethods: at least one judge is required");
@@ -5899,18 +5381,13 @@ function minimumBootstrapResamples(confidence, comparisonCount) {
5899
5381
  }
5900
5382
  function createOptimizationMethodInput(opts, methodName, resolvedRunDir, seed, baselineSurface, costLedger, optimizationSignal) {
5901
5383
  const methodRunDir = `${resolvedRunDir}/optimization/${slug(methodName)}`;
5902
- const cloneScenarios = (scenarios) => Object.freeze(scenarios.map((scenario) => structuredClone(scenario)));
5903
- const judges = opts.judges.map((judge) => Object.freeze({
5904
- ...judge,
5905
- dimensions: Object.freeze(judge.dimensions.map((dimension) => Object.freeze({ ...dimension })))
5906
- }));
5907
5384
  const signal = combineAbortSignals(opts.signal, opts.optimizationRunOptions?.signal, optimizationSignal);
5908
5385
  return Object.freeze({
5909
5386
  baselineSurface: structuredClone(baselineSurface),
5910
- trainScenarios: cloneScenarios(opts.trainScenarios),
5911
- selectionScenarios: cloneScenarios(opts.selectionScenarios),
5387
+ trainScenarios: opts.trainScenarios,
5388
+ selectionScenarios: opts.selectionScenarios,
5912
5389
  dispatchWithSurface: opts.dispatchWithSurface,
5913
- judges: Object.freeze(judges),
5390
+ judges: opts.judges,
5914
5391
  runDir: methodRunDir,
5915
5392
  seed,
5916
5393
  runOptions: Object.freeze({
@@ -6440,6 +5917,6 @@ function firstString(value) {
6440
5917
  return typeof value === "string" && value.trim() ? value : void 0;
6441
5918
  }
6442
5919
  //#endregion
6443
- export { runCanaries as $, recordCandidatePopulationSearch as A, buildReflectionPrompt as B, provenanceSpansPath as C, computeManifestHash as Ct, isProposedCandidate as D, runOptimization as E, searchHistoryCoverageRow as F, paretoFrontierWithCrowding as G, recoverTruncatedJson as H, verifySearchHistoryReceipt as I, defaultProductionGate as J, runFinalComparison as K, campaignBreakdown as L, assertCompleteSearchHistory as M, assertSearchHistoryMatchesReplay as N, labelTrustRank as O, createSearchHistoryReceipt as P, scoreRedTeamOutput as Q, campaignMeanComposite as R, provenanceRecordPath as S, cellCachePath as St, runImprovementLoop as T, dominates as U, parseReflectionResponse as V, paretoFrontier as W, redTeamDataset as X, DEFAULT_RED_TEAM_CORPUS as Y, redTeamReport as Z, campaignMeasurementDigest as _, campaignScenarioIdentity as _t, transientDispatchFailure as a, surfaceDispatchRef as at, loopProvenanceArgsFromResult as b, readCachedCell as bt, assertOptimizationResult as c, runCampaign as ct, assertGepaCandidatePopulationSummary as d, tangleTracesRoot as dt, assertCodeSurfaceIdentity as et, readGepaCandidatePopulationArtifact as f, BackendIntegrityError as ft, buildLoopProvenanceRecord as g, assertCampaignSplitIdentity as gt, createMethodCostScope as h, assertCampaignDesign as ht, quotaExhaustedUntil as i, surfaceContentHash as it, SearchHistoryRequiredError as j, SearchRecorder as k, compareOptimizationMethods as l, planCampaignRun as lt, costFromLedgerSummary as m, summarizeBackendIntegrity as mt, JudgeParseError as n, componentSurfaceIdentityMaterial as nt, aggregateRunScore as o, surfaceHash as ot, combineComparisonCosts as p, assertRealBackend as pt, openAutoPr as q, isTransientTransportFailure as r, renderSurfaceDiff as rt, clamp01 as s, runEval as st, llmJudge as t, codeSurfaceIdentityMaterial as tt, optimizationTokenUsageFromSummary as u, resolveRunDir as ut, canonicalDigest as v, campaignSplitDigest as vt, verifyLoopProvenanceRecord as w, loopProvenanceSpans as x, buildCellSchedule as xt, emitLoopProvenance as y, campaignSplitDigestFromIdentities as yt, compareRankKeys as z };
5920
+ export { openSearchLedger as A, redTeamDataset as B, assertSearchHistoryAdmissionOptions as C, verifySearchHistoryArtifact as D, searchHistoryCoverageRow as E, paretoFrontierWithCrowding as F, runCampaign as G, scoreRedTeamOutput as H, runFinalComparison as I, tangleTracesRoot as J, planCampaignRun as K, openAutoPr as L, validateSearchLedgerEvent as M, dominates as N, verifySearchHistoryReceipt as O, paretoFrontier as P, computeManifestHash as Q, defaultProductionGate as R, assertCompleteSearchHistory as S, createSearchHistoryReceipt as T, runCanaries as U, redTeamReport as V, runEval as W, buildCellSchedule as X, readCachedCell as Y, cellCachePath as Z, isProposedCandidate as _, transientDispatchFailure as a, recordCandidatePopulationSearch as b, compareOptimizationMethods as c, combineComparisonCosts as d, costFromLedgerSummary as f, runOptimization as g, runImprovementLoop as h, quotaExhaustedUntil as i, replaySearchLedgerText as j, FileSearchLedger as k, optimizationTokenUsageFromSummary as l, readGepaCandidatePopulationArtifact as m, JudgeParseError as n, aggregateRunScore as o, assertGepaCandidatePopulationSummary as p, resolveRunDir as q, isTransientTransportFailure as r, clamp01 as s, llmJudge as t, executeOptimizationMethod as u, labelTrustRank as v, assertSearchHistoryMatchesReplay as w, SearchHistoryRequiredError as x, SearchRecorder as y, DEFAULT_RED_TEAM_CORPUS as z };
6444
5921
 
6445
- //# sourceMappingURL=llm-judge-DmNaBrXB.js.map
5922
+ //# sourceMappingURL=llm-judge-DliimmRb.js.map