@tangle-network/agent-eval 0.174.0 → 0.176.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/CHANGELOG.md +37 -0
  2. package/README.md +1 -1
  3. package/dist/adapters/http.d.ts +1 -1
  4. package/dist/agent-profile-cell-0gSi5ffD.js +374 -0
  5. package/dist/agent-profile-cell-0gSi5ffD.js.map +1 -0
  6. package/dist/analyst/index.d.ts +3 -3
  7. package/dist/analyst/index.js +4 -4
  8. package/dist/{benchmark-command-mZIlR-ra.js → benchmark-command-yPqjcZnC.js} +7 -7
  9. package/dist/{benchmark-command-mZIlR-ra.js.map → benchmark-command-yPqjcZnC.js.map} +1 -1
  10. package/dist/benchmarks/index.d.ts +1 -1
  11. package/dist/benchmarks/index.js +3 -3
  12. package/dist/campaign/index.d.ts +3 -3
  13. package/dist/campaign/index.js +9 -9
  14. package/dist/{campaign-BzMSCejE.js → campaign-85igdlgG.js} +12 -12
  15. package/dist/{campaign-BzMSCejE.js.map → campaign-85igdlgG.js.map} +1 -1
  16. package/dist/campaign-evidence-D8DBLqLI.js +2083 -0
  17. package/dist/campaign-evidence-D8DBLqLI.js.map +1 -0
  18. package/dist/{opencode-sqlite-eK6HW6dr.js → claude-jsonl-CxZZrDJ3.js} +9 -149
  19. package/dist/claude-jsonl-CxZZrDJ3.js.map +1 -0
  20. package/dist/cli.js +9 -2
  21. package/dist/cli.js.map +1 -1
  22. package/dist/contract/index.d.ts +4 -4
  23. package/dist/contract/index.js +9 -9
  24. package/dist/{default-registry-CrAp0pYq.js → default-registry-DBqVI4pq.js} +2 -2
  25. package/dist/{default-registry-CrAp0pYq.js.map → default-registry-DBqVI4pq.js.map} +1 -1
  26. package/dist/{define-agent-eval-V1jQyCDR.d.ts → define-agent-eval-CCbl8k2E.d.ts} +11 -4
  27. package/dist/define-agent-eval-CCbl8k2E.d.ts.map +1 -0
  28. package/dist/{define-agent-eval-ox5McL6e.js → define-agent-eval-DEMsu5eA.js} +54 -36
  29. package/dist/define-agent-eval-DEMsu5eA.js.map +1 -0
  30. package/dist/{dspy-rlm-engine-Caz2pl4L.js → dspy-rlm-engine-DqjER2sV.js} +2 -2
  31. package/dist/{dspy-rlm-engine-Caz2pl4L.js.map → dspy-rlm-engine-DqjER2sV.js.map} +1 -1
  32. package/dist/{eval-campaign-BeAjdhzC.js → eval-campaign-Cs-7MiCs.js} +4 -5
  33. package/dist/{eval-campaign-BeAjdhzC.js.map → eval-campaign-Cs-7MiCs.js.map} +1 -1
  34. package/dist/experiment/index.d.ts +3 -68
  35. package/dist/experiment/index.d.ts.map +1 -1
  36. package/dist/experiment/index.js +6 -128
  37. package/dist/experiment/index.js.map +1 -1
  38. package/dist/{attestation-XSUpbc4o.js → experiment-tracker-BKEumQug.js} +2 -96
  39. package/dist/experiment-tracker-BKEumQug.js.map +1 -0
  40. package/dist/{attestation-c1QvaBdX.d.ts → experiment-tracker-CNwqCZFD.d.ts} +2 -78
  41. package/dist/experiment-tracker-CNwqCZFD.d.ts.map +1 -0
  42. package/dist/{external-optimizer-process-CxnFL1hd.js → external-optimizer-process-Dlz8YxrT.js} +3 -3
  43. package/dist/{external-optimizer-process-CxnFL1hd.js.map → external-optimizer-process-Dlz8YxrT.js.map} +1 -1
  44. package/dist/{external-optimizer-subprocess-CQi27uEI.js → external-optimizer-subprocess-q3VzlGAO.js} +2 -2
  45. package/dist/{external-optimizer-subprocess-CQi27uEI.js.map → external-optimizer-subprocess-q3VzlGAO.js.map} +1 -1
  46. package/dist/{index-Bn-nlnSV.d.ts → index-BAAiSF3_.d.ts} +2 -2
  47. package/dist/{index-Bn-nlnSV.d.ts.map → index-BAAiSF3_.d.ts.map} +1 -1
  48. package/dist/{index-DKXuBPXf.d.ts → index-Bg6OT2Dd.d.ts} +23 -10
  49. package/dist/{index-DKXuBPXf.d.ts.map → index-Bg6OT2Dd.d.ts.map} +1 -1
  50. package/dist/{index-BTrx5s8m.d.ts → index-DBkcm_9H.d.ts} +4 -4
  51. package/dist/{index-BTrx5s8m.d.ts.map → index-DBkcm_9H.d.ts.map} +1 -1
  52. package/dist/{index-D-UdhAmg.d.ts → index-u0d1Jp4F.d.ts} +4 -2
  53. package/dist/{index-D-UdhAmg.d.ts.map → index-u0d1Jp4F.d.ts.map} +1 -1
  54. package/dist/index.d.ts +5 -5
  55. package/dist/index.js +15 -16
  56. package/dist/index.js.map +1 -1
  57. package/dist/{integrity-BWywb34E.js → integrity-DsHWCebQ.js} +11 -435
  58. package/dist/integrity-DsHWCebQ.js.map +1 -0
  59. package/dist/ledger-core/index.d.ts +2 -2
  60. package/dist/ledger-core/index.js +2 -2
  61. package/dist/{ledger-core-PIfjCbKn.js → ledger-core-Cs9f7385.js} +60 -47
  62. package/dist/{ledger-core-PIfjCbKn.js.map → ledger-core-Cs9f7385.js.map} +1 -1
  63. package/dist/{llm-judge-DmNaBrXB.js → llm-judge-DliimmRb.js} +994 -1517
  64. package/dist/llm-judge-DliimmRb.js.map +1 -0
  65. package/dist/{mint-vWOdD8Ae.js → mint-Cc1_zwRQ.js} +2 -2
  66. package/dist/{mint-vWOdD8Ae.js.map → mint-Cc1_zwRQ.js.map} +1 -1
  67. package/dist/openapi.json +1 -1
  68. package/dist/opencode-sqlite-CNw3vubS.js +145 -0
  69. package/dist/opencode-sqlite-CNw3vubS.js.map +1 -0
  70. package/dist/{produced-state-B8mw6zj9.js → produced-state-DrMqa2HD.js} +3 -2
  71. package/dist/{produced-state-B8mw6zj9.js.map → produced-state-DrMqa2HD.js.map} +1 -1
  72. package/dist/profile-cell.js +1 -268
  73. package/dist/{promotion-policy-LY9mVQ7W.js → promotion-policy-DWOm70gx.js} +2 -2
  74. package/dist/{promotion-policy-LY9mVQ7W.js.map → promotion-policy-DWOm70gx.js.map} +1 -1
  75. package/dist/{release-confidence-BsGEg_xg.js → release-confidence-BcGCclTB.js} +2 -2
  76. package/dist/{release-confidence-BsGEg_xg.js.map → release-confidence-BcGCclTB.js.map} +1 -1
  77. package/dist/report-command-DKlXfU5r.js +1528 -0
  78. package/dist/report-command-DKlXfU5r.js.map +1 -0
  79. package/dist/reporting.js +2 -2
  80. package/dist/{reward-hacking-CKW4teig.js → reward-hacking-D0XwhVWE.js} +2 -215
  81. package/dist/reward-hacking-D0XwhVWE.js.map +1 -0
  82. package/dist/rl.js +5 -4
  83. package/dist/rl.js.map +1 -1
  84. package/dist/rollout/index.js +4 -3
  85. package/dist/{rollout-C-znbbYg.js → rollout-DmoJVqrF.js} +4 -3
  86. package/dist/{rollout-C-znbbYg.js.map → rollout-DmoJVqrF.js.map} +1 -1
  87. package/dist/run-record-CR63CpHK.js +216 -0
  88. package/dist/run-record-CR63CpHK.js.map +1 -0
  89. package/dist/{run-record-ZIsR9Fif.js → run-record-DQpSf7t-.js} +2 -2
  90. package/dist/{run-record-ZIsR9Fif.js.map → run-record-DQpSf7t-.js.map} +1 -1
  91. package/dist/{semantic-concept-judge-E3s_fEjB.js → semantic-concept-judge-Dw-f7TEs.js} +3 -3
  92. package/dist/{semantic-concept-judge-E3s_fEjB.js.map → semantic-concept-judge-Dw-f7TEs.js.map} +1 -1
  93. package/dist/{sequential-B51qAYE4.js → sequential-B5gXgcyp.js} +3 -3
  94. package/dist/{sequential-B51qAYE4.js.map → sequential-B5gXgcyp.js.map} +1 -1
  95. package/dist/{skillopt-optimization-method-f7399oGb.js → skillopt-optimization-method-CV7go7ex.js} +6 -749
  96. package/dist/skillopt-optimization-method-CV7go7ex.js.map +1 -0
  97. package/dist/{statistical-heldout-Cqb73yE9.d.ts → statistical-heldout-Z9NROFFS.d.ts} +156 -3
  98. package/dist/statistical-heldout-Z9NROFFS.d.ts.map +1 -0
  99. package/dist/{summary-report-Bgh8CpNK.js → summary-report-B16xy9Kd.js} +2 -2
  100. package/dist/{summary-report-Bgh8CpNK.js.map → summary-report-B16xy9Kd.js.map} +1 -1
  101. package/dist/supervisor-run/index.d.ts +71 -6
  102. package/dist/supervisor-run/index.d.ts.map +1 -1
  103. package/dist/supervisor-run/index.js +6 -1357
  104. package/dist/supervisor-run/index.js.map +1 -1
  105. package/dist/terminal-record-Ce9_UjRz.js +539 -0
  106. package/dist/terminal-record-Ce9_UjRz.js.map +1 -0
  107. package/dist/traces.js +1 -1
  108. package/dist/{types-CoPUTiXb.d.ts → types-vUdAx2Cj.d.ts} +65 -3
  109. package/dist/types-vUdAx2Cj.d.ts.map +1 -0
  110. package/docs/public-api.md +62 -39
  111. package/docs/search-history-receipts.md +48 -1
  112. package/package.json +1 -1
  113. package/dist/attestation-XSUpbc4o.js.map +0 -1
  114. package/dist/attestation-c1QvaBdX.d.ts.map +0 -1
  115. package/dist/define-agent-eval-V1jQyCDR.d.ts.map +0 -1
  116. package/dist/define-agent-eval-ox5McL6e.js.map +0 -1
  117. package/dist/integrity-BWywb34E.js.map +0 -1
  118. package/dist/llm-judge-DmNaBrXB.js.map +0 -1
  119. package/dist/opencode-sqlite-eK6HW6dr.js.map +0 -1
  120. package/dist/power-preflight-CFXm0Vjo.js +0 -502
  121. package/dist/power-preflight-CFXm0Vjo.js.map +0 -1
  122. package/dist/pre-registration-D94b7Of5.js +0 -110
  123. package/dist/pre-registration-D94b7Of5.js.map +0 -1
  124. package/dist/profile-cell.js.map +0 -1
  125. package/dist/reward-hacking-CKW4teig.js.map +0 -1
  126. package/dist/skillopt-optimization-method-f7399oGb.js.map +0 -1
  127. package/dist/statistical-heldout-Cqb73yE9.d.ts.map +0 -1
  128. package/dist/types-CoPUTiXb.d.ts.map +0 -1
@@ -0,0 +1,2083 @@
1
+ import { t as AgentEvalError } from "./errors-Dngq5h35.js";
2
+ import { a as hashCanonical, i as compareCodeUnits, r as canonicalString } from "./canonical-DPyQ_rpt.js";
3
+ import { g as pairedRiskDifferenceScore, h as pairedRiskDifferenceExact, p as pairedBinaryScale } from "./paired-arms-D4aeIHUy.js";
4
+ import { a as pairedSignTest, i as pairedDeltaTieFraction, r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
5
+ import { n as contentHash } from "./verdict-cache-B3eCVQtY.js";
6
+ import { n as campaignCellExecutionEvidence, o as projectCampaignCellQuality } from "./run-record-CR63CpHK.js";
7
+ import { createHash } from "node:crypto";
8
+ import { join } from "node:path";
9
+ //#region src/campaign/coverage.ts
10
+ /** Reject campaign designs whose denominator cannot be identified exactly. */
11
+ function assertCampaignDesign(scenarios, reps) {
12
+ if (!Number.isSafeInteger(reps) || reps < 1) throw new Error("campaign design requires reps to be a positive safe integer");
13
+ const scenarioIds = /* @__PURE__ */ new Set();
14
+ for (const scenario of scenarios) {
15
+ if (typeof scenario.id !== "string" || scenario.id.trim().length === 0) throw new Error("campaign design requires every scenario to have a non-empty id");
16
+ if (scenarioIds.has(scenario.id)) throw new Error(`campaign design contains duplicate scenario id '${scenario.id}'`);
17
+ if (typeof scenario.kind !== "string" || scenario.kind.trim().length === 0) throw new Error("campaign design requires every scenario to have a non-empty kind");
18
+ if (scenario.seedGroup !== void 0 && (typeof scenario.seedGroup !== "string" || scenario.seedGroup.trim().length === 0)) throw new Error("campaign design requires seedGroup to be a non-empty string when set");
19
+ scenarioIds.add(scenario.id);
20
+ }
21
+ }
22
+ /** Redacted but independently verifiable identity of one complete scenario. */
23
+ function campaignScenarioIdentity(scenario) {
24
+ assertCampaignDesign([scenario], 1);
25
+ return {
26
+ id: scenario.id,
27
+ kind: scenario.kind,
28
+ scenarioDigest: `sha256:${contentHash(scenario)}`
29
+ };
30
+ }
31
+ /** Canonical split identity reconstructed from redacted scenario identities. */
32
+ function campaignSplitDigestFromIdentities(scenarios, reps) {
33
+ assertCampaignDesign(scenarios, reps);
34
+ for (const scenario of scenarios) if (!/^sha256:[a-f0-9]{64}$/.test(scenario.scenarioDigest)) throw new Error(`campaign scenario '${scenario.id}' has an invalid digest`);
35
+ return `sha256:${contentHash({
36
+ schema: "tangle.campaign-split",
37
+ scenarios: scenarios.map(({ id, kind, scenarioDigest }) => ({
38
+ id,
39
+ kind,
40
+ scenarioDigest
41
+ })),
42
+ reps
43
+ })}`;
44
+ }
45
+ /** Canonical identity of the exact scenario payloads and replicate count. */
46
+ function campaignSplitDigest(scenarios, reps) {
47
+ assertCampaignDesign(scenarios, reps);
48
+ return campaignSplitDigestFromIdentities(scenarios.map(campaignScenarioIdentity), reps);
49
+ }
50
+ /** Refuse a campaign whose retained task identities contradict its split digest. */
51
+ function assertCampaignSplitIdentity(scenarios, reps, splitDigest) {
52
+ if (campaignSplitDigestFromIdentities(scenarios, reps) !== splitDigest) throw new Error("campaign split digest does not match its retained scenario identities");
53
+ }
54
+ /** Exact designed-denominator receipt for one campaign. */
55
+ function campaignCoverage(cells, scenarios, reps, requireJudgeScore) {
56
+ assertCampaignDesign(scenarios, reps);
57
+ const expectedCellIds = designedCellIds(scenarios, reps);
58
+ const cellsById = /* @__PURE__ */ new Map();
59
+ for (const cell of cells) {
60
+ const matches = cellsById.get(cell.cellId) ?? [];
61
+ matches.push(cell);
62
+ cellsById.set(cell.cellId, matches);
63
+ }
64
+ const scorableCellIds = [];
65
+ const unscorableCells = [];
66
+ for (const cellId of expectedCellIds) {
67
+ const matches = cellsById.get(cellId) ?? [];
68
+ if (matches.length === 0) {
69
+ unscorableCells.push({
70
+ cellId,
71
+ reason: "missing campaign cell"
72
+ });
73
+ continue;
74
+ }
75
+ if (matches.length > 1) {
76
+ unscorableCells.push({
77
+ cellId,
78
+ reason: `duplicate campaign cell (${matches.length})`
79
+ });
80
+ continue;
81
+ }
82
+ const cell = matches[0];
83
+ const scoreEntries = Object.entries(cell.judgeScores);
84
+ const successfulScores = scoreEntries.map(([, score]) => score).filter((score) => score.failed !== true && Number.isFinite(score.composite));
85
+ const nonFiniteScores = scoreEntries.filter(([, score]) => score.failed !== true && (!Number.isFinite(score.composite) || Object.values(score.dimensions).some((value) => !Number.isFinite(value))));
86
+ const reasons = [];
87
+ if (cell.error) reasons.push(cell.error);
88
+ if (cell.artifact === null || cell.artifact === void 0) reasons.push("missing artifact");
89
+ if (!cell.error && requireJudgeScore && successfulScores.length === 0) reasons.push("no successful finite judge score");
90
+ if (scoreEntries.some(([, score]) => score.failed === true)) reasons.push("judge score marked failed");
91
+ const failedPanelJudges = [...new Set(scoreEntries.flatMap(([, score]) => score.failedJudges ?? []))].sort();
92
+ if (failedPanelJudges.length > 0) reasons.push(`judge panel incomplete: ${failedPanelJudges.join(", ")}`);
93
+ if (nonFiniteScores.length > 0) reasons.push(`non-finite judge score: ${nonFiniteScores.map(([name]) => name).sort().join(", ")}`);
94
+ if (reasons.length > 0) unscorableCells.push({
95
+ cellId,
96
+ reason: reasons.join("; ")
97
+ });
98
+ else scorableCellIds.push(cellId);
99
+ }
100
+ const expected = new Set(expectedCellIds);
101
+ for (const cell of cells) {
102
+ if (cell.cellId !== `${cell.scenarioId}:${cell.rep}`) {
103
+ unscorableCells.push({
104
+ cellId: cell.cellId,
105
+ reason: "campaign cell id does not match scenario id and rep"
106
+ });
107
+ continue;
108
+ }
109
+ if (!expected.has(cell.cellId)) unscorableCells.push({
110
+ cellId: cell.cellId,
111
+ reason: "unexpected campaign cell"
112
+ });
113
+ }
114
+ return {
115
+ complete: unscorableCells.length === 0 && scorableCellIds.length === expectedCellIds.length,
116
+ expectedCellIds,
117
+ scorableCellIds,
118
+ unscorableCells
119
+ };
120
+ }
121
+ function formatCoverageFailures(coverage) {
122
+ const shown = coverage.unscorableCells.slice(0, 3).map((cell) => `${cell.cellId}: ${cell.reason}`).join("; ");
123
+ const remainder = coverage.unscorableCells.length - Math.min(3, coverage.unscorableCells.length);
124
+ return remainder > 0 ? `${shown}; +${remainder} more` : shown || "unknown coverage failure";
125
+ }
126
+ function designedCellIds(scenarios, reps) {
127
+ const ids = [];
128
+ for (const scenario of scenarios) for (let rep = 0; rep < reps; rep++) ids.push(`${scenario.id}:${rep}`);
129
+ return ids;
130
+ }
131
+ /** Require the complete designed denominator before a final comparison. */
132
+ function assertCompleteCampaign(campaign, scenarios, reps, requireJudgeScore, label) {
133
+ const coverage = campaignCoverage(campaign.cells, scenarios, reps, requireJudgeScore);
134
+ if (!coverage.complete) throw new Error(`${label} is incomplete (${coverage.scorableCellIds.length}/${coverage.expectedCellIds.length} designed cells scorable) — ${formatCoverageFailures(coverage)}. Refusing to compare unequal results.`);
135
+ }
136
+ //#endregion
137
+ //#region src/integrity/backend-integrity.ts
138
+ /**
139
+ * Error thrown when an integrity assertion fails. Caller can pattern-match
140
+ * by `code === 'AGENT_EVAL_BACKEND_STUB'` to differentiate from other
141
+ * errors.
142
+ */
143
+ var BackendIntegrityError = class extends AgentEvalError {
144
+ report;
145
+ constructor(message, report) {
146
+ super("backend_integrity", message);
147
+ this.report = report;
148
+ }
149
+ };
150
+ /**
151
+ * Inspect a batch of RunRecords and return an integrity report. Pure
152
+ * function — no I/O, no logging. The caller decides what to do with the
153
+ * verdict (print warning, throw, gate CI, etc.).
154
+ */
155
+ function summarizeBackendIntegrity(records) {
156
+ return summarizeBackendUsage(records.map((record) => ({
157
+ inputTokens: record.tokenUsage.input,
158
+ outputTokens: record.tokenUsage.output,
159
+ costUsd: record.costUsd
160
+ })));
161
+ }
162
+ /** Inspect settled agent calls from the canonical cost ledger. */
163
+ function summarizeAgentReceiptIntegrity(receipts) {
164
+ return summarizeBackendUsage(receipts.filter((receipt) => receipt.channel === "agent").map((receipt) => ({
165
+ inputTokens: receipt.inputTokens,
166
+ outputTokens: receipt.outputTokens,
167
+ costUsd: receipt.costUsd
168
+ })));
169
+ }
170
+ function summarizeBackendUsage(records) {
171
+ const totalRecords = records.length;
172
+ let stubRecords = 0;
173
+ let realRecords = 0;
174
+ let uncostedRecords = 0;
175
+ let totalInputTokens = 0;
176
+ let totalOutputTokens = 0;
177
+ let totalCostUsd = 0;
178
+ for (const rec of records) {
179
+ totalInputTokens += rec.inputTokens;
180
+ totalOutputTokens += rec.outputTokens;
181
+ totalCostUsd += rec.costUsd ?? 0;
182
+ if (rec.inputTokens === 0 && rec.outputTokens === 0) stubRecords++;
183
+ else realRecords++;
184
+ if (rec.outputTokens > 0 && (rec.costUsd === null || rec.costUsd === 0)) uncostedRecords++;
185
+ }
186
+ const verdict = totalRecords === 0 ? "stub" : stubRecords === totalRecords ? "stub" : stubRecords === 0 ? "real" : "mixed";
187
+ const diagnosis = buildDiagnosis({
188
+ totalRecords,
189
+ stubRecords,
190
+ realRecords,
191
+ uncostedRecords,
192
+ totalInputTokens,
193
+ totalOutputTokens,
194
+ totalCostUsd,
195
+ verdict
196
+ });
197
+ return {
198
+ totalRecords,
199
+ stubRecords,
200
+ realRecords,
201
+ uncostedRecords,
202
+ totalInputTokens,
203
+ totalOutputTokens,
204
+ totalCostUsd,
205
+ verdict,
206
+ diagnosis
207
+ };
208
+ }
209
+ function buildDiagnosis(r) {
210
+ if (r.totalRecords === 0) return "no records — eval produced zero runs; backend likely failed before first turn";
211
+ if (r.verdict === "stub") return [
212
+ `all ${r.totalRecords} records have zero token usage — the LLM backend was never called.`,
213
+ "common causes: --backend sandbox without a sandbox bridge running; stub model returning hard-coded strings;",
214
+ "auth misconfigured so requests were silently dropped before the LLM. Re-run with --backend tcloud and TANGLE_API_KEY set,",
215
+ "or boot the cli-bridge / sandbox before invoking the eval."
216
+ ].join(" ");
217
+ if (r.verdict === "mixed") {
218
+ const pct = (r.stubRecords / r.totalRecords * 100).toFixed(0);
219
+ return [
220
+ `${r.stubRecords}/${r.totalRecords} records (${pct}%) have zero token usage — the backend partially failed.`,
221
+ "common causes: rate-limit cascade (429s after the first N personas);",
222
+ "transient auth expiry mid-run; provider outage. Treat the affected records as missing data, not agent failures."
223
+ ].join(" ");
224
+ }
225
+ if (r.uncostedRecords > 0) {
226
+ const pct = (r.uncostedRecords / r.totalRecords * 100).toFixed(0);
227
+ return [
228
+ `${r.totalRecords} records with real LLM activity (in=${r.totalInputTokens}, out=${r.totalOutputTokens} tokens).`,
229
+ `${r.uncostedRecords} (${pct}%) have output tokens but costUsd=0. Two distinct roots:`,
230
+ "(a) cost ledger mis-wired — no usage propagation from the runtime stream into RunRecord; or",
231
+ "(b) the model is unpriced at the source (sandbox/router returned $0 despite real tokens).",
232
+ "For (b), price the measured tokens against the substrate table (estimateCost) instead of leaving $0."
233
+ ].join(" ");
234
+ }
235
+ return `${r.totalRecords} records with real LLM activity (in=${r.totalInputTokens}, out=${r.totalOutputTokens} tokens, $${r.totalCostUsd.toFixed(4)}).`;
236
+ }
237
+ /**
238
+ * Throw BackendIntegrityError if the verdict is 'stub' — i.e. every record
239
+ * shows zero LLM activity. Non-strict callers can pass `{ allowMixed: false }`
240
+ * to also reject mixed verdicts (recommended for CI gates).
241
+ *
242
+ * Real backends pass through silently.
243
+ */
244
+ function assertRealBackend(records, opts = {}) {
245
+ return assertBackendReport(summarizeBackendIntegrity(records), opts);
246
+ }
247
+ function assertBackendReport(report, opts) {
248
+ const allowMixed = opts.allowMixed ?? true;
249
+ if (report.verdict === "stub") throw new BackendIntegrityError(`backend-integrity: ran against a stub or unconfigured backend — ${report.diagnosis}`, report);
250
+ if (!allowMixed && report.verdict === "mixed") throw new BackendIntegrityError(`backend-integrity: partial backend failure rejected — ${report.diagnosis}`, report);
251
+ return report;
252
+ }
253
+ //#endregion
254
+ //#region src/campaign/surface-identity.ts
255
+ const GIT_OBJECT_ID = /^(?:[a-f0-9]{40}|[a-f0-9]{64})$/;
256
+ const SHA256 = /^sha256:[a-f0-9]{64}$/;
257
+ /** Validate the immutable identity shape; the owning executor verifies the Git objects and patch. */
258
+ function assertCodeSurfaceIdentity(surface) {
259
+ if (!surface || typeof surface !== "object") throw new TypeError("CodeSurface must be an object");
260
+ const candidate = surface;
261
+ if (candidate.kind !== "code") throw new TypeError("CodeSurface.kind must be \"code\"");
262
+ if (typeof candidate.worktreeRef !== "string" || candidate.worktreeRef.trim().length === 0) throw new TypeError("CodeSurface.worktreeRef must be a non-empty locator");
263
+ if (typeof candidate.baseRef !== "string" || candidate.baseRef.trim().length === 0) throw new TypeError("CodeSurface.baseRef must be a non-empty ref label");
264
+ for (const [field, value] of [
265
+ ["baseCommit", candidate.baseCommit],
266
+ ["baseTree", candidate.baseTree],
267
+ ["candidateCommit", candidate.candidateCommit],
268
+ ["candidateTree", candidate.candidateTree]
269
+ ]) if (typeof value !== "string" || !GIT_OBJECT_ID.test(value)) throw new TypeError(`CodeSurface.${field} must be a full Git object id`);
270
+ const patch = candidate.patch;
271
+ if (!patch || typeof patch !== "object" || patch.format !== "git-diff-binary") throw new TypeError("CodeSurface.patch.format must be \"git-diff-binary\"");
272
+ if (typeof patch.sha256 !== "string" || !SHA256.test(patch.sha256)) throw new TypeError("CodeSurface.patch.sha256 must be a sha256 digest");
273
+ if (!Number.isSafeInteger(patch.byteLength) || patch.byteLength < 0) throw new TypeError("CodeSurface.patch.byteLength must be a non-negative safe integer");
274
+ }
275
+ /** Assert that a value is a valid non-empty component surface. */
276
+ function assertComponentSurface(surface) {
277
+ if (!surface || typeof surface !== "object") throw new TypeError("ComponentSurface must be an object");
278
+ const candidate = surface;
279
+ if (candidate.kind !== "components") throw new TypeError("ComponentSurface.kind must be \"components\"");
280
+ if (!candidate.components || typeof candidate.components !== "object" || Array.isArray(candidate.components)) throw new TypeError("ComponentSurface.components must be an object");
281
+ const entries = Object.entries(candidate.components);
282
+ if (entries.length === 0) throw new TypeError("ComponentSurface.components must not be empty");
283
+ for (const [name, content] of entries) {
284
+ if (!name.trim() || name.trim() !== name) throw new TypeError("ComponentSurface component names must be trimmed and non-empty");
285
+ if (typeof content !== "string") throw new TypeError(`ComponentSurface component '${name}' must be a string`);
286
+ }
287
+ }
288
+ /**
289
+ * Deterministic identity material for a component surface.
290
+ *
291
+ * `canonicalString` orders keys by UTF-16 code unit (RFC 8785), which is a
292
+ * property of the value alone. The previous material ordered them with
293
+ * `localeCompare`, which reads the host's collation — so the same surface
294
+ * could produce two different identities on two machines, and the stored
295
+ * identity would stop matching a recomputation of the identical surface.
296
+ */
297
+ function componentSurfaceIdentityMaterial(surface) {
298
+ assertComponentSurface(surface);
299
+ return canonicalString({
300
+ schema: "tangle.component-surface",
301
+ components: surface.components
302
+ });
303
+ }
304
+ /**
305
+ * The retired material builder, kept PRIVATE and reachable only from
306
+ * {@link surfaceHashMatches}.
307
+ *
308
+ * A surface identity recorded before this release was minted from these bytes.
309
+ * The verify path tries the current material first and falls back to this one,
310
+ * so a stored identity still matches its own surface; nothing mints from it.
311
+ *
312
+ * Every component surface's identity moves, not only one whose names sort
313
+ * differently under the host's collation: RFC 8785 also orders the two
314
+ * top-level keys, so `components` precedes `schema` where this builder emitted
315
+ * them in literal order. The retention window therefore covers every stored
316
+ * component-surface identity, which is why this builder is kept rather than
317
+ * scoped to the mixed-case case.
318
+ */
319
+ function retiredComponentSurfaceIdentityMaterial(surface) {
320
+ assertComponentSurface(surface);
321
+ return JSON.stringify({
322
+ schema: "tangle.component-surface",
323
+ components: Object.fromEntries(Object.entries(surface.components).sort(([left], [right]) => left.localeCompare(right)))
324
+ });
325
+ }
326
+ /** Canonical, location-independent identity of a finalized code candidate.
327
+ * Commit metadata is excluded: two commits with the same base, final tree,
328
+ * and patch bytes are the same executable candidate. */
329
+ function codeSurfaceIdentityMaterial(surface) {
330
+ assertCodeSurfaceIdentity(surface);
331
+ return JSON.stringify({
332
+ schema: "tangle.code-surface",
333
+ baseCommit: surface.baseCommit,
334
+ baseTree: surface.baseTree,
335
+ candidateTree: surface.candidateTree,
336
+ patch: {
337
+ format: surface.patch.format,
338
+ sha256: surface.patch.sha256,
339
+ byteLength: surface.patch.byteLength
340
+ }
341
+ });
342
+ }
343
+ /** Full SHA-256 content identity for a prompt or finalized code surface. */
344
+ function surfaceContentHash(surface) {
345
+ const material = typeof surface === "string" ? surface : surface.kind === "components" ? componentSurfaceIdentityMaterial(surface) : codeSurfaceIdentityMaterial(surface);
346
+ return `sha256:${createHash("sha256").update(material).digest("hex")}`;
347
+ }
348
+ /** Short loop key derived from the same content identity as provenance. */
349
+ function surfaceHash(surface) {
350
+ return surfaceContentHash(surface).slice(7, 23);
351
+ }
352
+ /**
353
+ * Whether `storedHash` is the loop key of `surface`, under the current identity
354
+ * material or the retired one.
355
+ *
356
+ * A stored key is 16 hex characters with no room for a scheme tag, so the
357
+ * scheme cannot be read off the value the way an `agent-profile-cell` id names
358
+ * its own. The verify path therefore tries both, which gives the same property:
359
+ * a key minted by an earlier release still matches its own surface, and a
360
+ * surface that was actually edited matches neither.
361
+ *
362
+ * Only a component surface can differ between the two; a prompt or code surface
363
+ * produces identical material under both, so the second comparison is a no-op
364
+ * for them.
365
+ */
366
+ function surfaceHashMatches(surface, storedHash) {
367
+ if (surfaceHash(surface) === storedHash) return true;
368
+ if (typeof surface === "string" || surface.kind !== "components") return false;
369
+ return createHash("sha256").update(retiredComponentSurfaceIdentityMaterial(surface)).digest("hex").slice(0, 16) === storedHash;
370
+ }
371
+ /** Canonical customer-visible description of the exact before/after surfaces. */
372
+ function renderSurfaceDiff(winnerSurface, baselineSurface) {
373
+ if (typeof winnerSurface === "string" && typeof baselineSurface === "string") return [
374
+ "--- baseline",
375
+ "+++ winner",
376
+ ...baselineSurface.split("\n").map((line) => `- ${line}`),
377
+ ...winnerSurface.split("\n").map((line) => `+ ${line}`)
378
+ ].join("\n");
379
+ const describe = (surface) => {
380
+ if (typeof surface === "string") return "(prompt surface)";
381
+ if (surface.kind === "components") {
382
+ assertComponentSurface(surface);
383
+ return Object.entries(surface.components).sort(([left], [right]) => left.localeCompare(right)).map(([name, content]) => `[${name}]\n${content}`).join("\n\n");
384
+ }
385
+ assertCodeSurfaceIdentity(surface);
386
+ return [
387
+ `baseCommit=${surface.baseCommit}`,
388
+ `baseTree=${surface.baseTree}`,
389
+ `candidateCommit=${surface.candidateCommit}`,
390
+ `candidateTree=${surface.candidateTree}`,
391
+ `patch=${surface.patch.sha256}`,
392
+ `patchBytes=${surface.patch.byteLength}`,
393
+ ...surface.summary ? [surface.summary] : []
394
+ ].join("\n");
395
+ };
396
+ return `--- baseline\n${describe(baselineSurface)}\n+++ winner\n${describe(winnerSurface)}`;
397
+ }
398
+ /** Bind a campaign cache entry to the exact surface and caller-owned execution revision. */
399
+ function surfaceDispatchRef(surface, executionRef = "anonymous") {
400
+ if (!executionRef.trim() || executionRef.trim() !== executionRef) throw new Error("surfaceDispatchRef: executionRef must be trimmed and non-empty");
401
+ return `surface:${executionRef}:${surfaceContentHash(surface)}`;
402
+ }
403
+ //#endregion
404
+ //#region src/paired-delta-test.ts
405
+ /** Smallest all-positive sample that can clear a one-sided exact sign test. */
406
+ function minimumPairsForPairedDeltaTest(confidence = .95) {
407
+ if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
408
+ const oneSidedAlpha = (1 - confidence) / 2;
409
+ return Math.ceil(Math.log2(1 / oneSidedAlpha));
410
+ }
411
+ /**
412
+ * Tests whether a paired candidate-minus-baseline delta clears a threshold.
413
+ *
414
+ * At 20 or more pairs, the percentile bootstrap lower bound carries the
415
+ * decision. Below that point the interval is descriptive only, so the function
416
+ * switches to a pre-registered one-sided exact sign test. The exact path is
417
+ * deliberately conservative: it requires both a point estimate above the
418
+ * threshold and enough consistently positive paired differences.
419
+ *
420
+ * ## A zero-width interval is never significant
421
+ *
422
+ * When every paired delta is identical the resample distribution is a point
423
+ * mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n
424
+ * identical deltas of g. Neither says the effect is certain — both say the
425
+ * sample carries no information about how far the estimate could be wrong, and
426
+ * `low > threshold` then answers on the point estimate alone. It fails in both
427
+ * directions: `[0, 0]` clears every NEGATIVE threshold, which is how a
428
+ * tie-dominated pass/fail comparison laundered a regression into a
429
+ * noninferiority pass, and `[g, g]` clears every threshold below g with no
430
+ * spread behind it. Under a bounded asymmetric null whose true mean paired
431
+ * delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —
432
+ * every sample that misses the drop is exactly that shape, and deciding on
433
+ * `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.
434
+ *
435
+ * So `indeterminate` is reported and `significant` is false whenever the
436
+ * interval has zero width, on BOTH paths: at small n the exact sign test is a
437
+ * test of the MEDIAN and a zero-spread sample is precisely where it stops
438
+ * saying anything about the mean the caller is thresholding.
439
+ *
440
+ * `threshold` may be negative — that is a noninferiority margin, and it is the
441
+ * regime the zero-width hole is worst in. For a two-point (pass/fail) outcome
442
+ * the percentile bootstrap is not a valid interval at a nonzero margin at all;
443
+ * use {@link decidePairedPromotion}, which routes those to Tango's score
444
+ * interval, rather than thresholding this function's bootstrap directly.
445
+ */
446
+ function pairedDeltaTest(before, after, options = {}) {
447
+ const threshold = options.threshold ?? 0;
448
+ if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
449
+ const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
450
+ const requestedMinimum = options.minPairs ?? exactMinimum;
451
+ if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
452
+ const minimumPairs = Math.max(requestedMinimum, exactMinimum);
453
+ const bootstrap = pairedBootstrap(before, after, options);
454
+ const sufficient = bootstrap.n >= minimumPairs;
455
+ const indeterminate = bootstrap.n > 0 && (!Number.isFinite(bootstrap.low) || !Number.isFinite(bootstrap.high) || bootstrap.low === bootstrap.high);
456
+ if (bootstrap.gateEligible) return {
457
+ bootstrap,
458
+ method: "bootstrap-ci",
459
+ pValue: null,
460
+ minimumPairs,
461
+ sufficient,
462
+ indeterminate,
463
+ significant: sufficient && !indeterminate && bootstrap.low > threshold
464
+ };
465
+ const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
466
+ const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
467
+ return {
468
+ bootstrap,
469
+ method: "exact-sign",
470
+ pValue: exact.pValue,
471
+ minimumPairs,
472
+ sufficient,
473
+ indeterminate,
474
+ significant: sufficient && !indeterminate && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
475
+ };
476
+ }
477
+ //#endregion
478
+ //#region src/paired-promotion-decision.ts
479
+ /**
480
+ * @module
481
+ * ONE rule for "does this paired interval clear a promotion threshold".
482
+ *
483
+ * The rule below was derived on `HeldOutGate` (#479) after the same estimator
484
+ * bug shipped twice. It then turned out that a SECOND gate — the composable
485
+ * `heldOutGate`, plus everything else routed through `heldoutSignificance` —
486
+ * still carried the original defect, because the rule had been written into one
487
+ * gate's method body rather than into a shared function. Two copies of a
488
+ * statistical rule is how a defect survives in one of them, so there is now
489
+ * exactly one copy and both gates call it.
490
+ *
491
+ * Three things the rule does that a bare `pairedBootstrap(...).low > threshold`
492
+ * does not:
493
+ *
494
+ * 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a
495
+ * pass/fail eval the paired delta vector is dominated by ties, so the
496
+ * bootstrap of the mean is a resample of a lattice with three atoms and its
497
+ * percentile interval is not valid at a nonzero margin. The score interval
498
+ * (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under
499
+ * each hypothesised margin instead of fixing it at the observed value, which
500
+ * is the only construction that stays a confidence interval as the margin
501
+ * moves off zero — the regime every noninferiority threshold lives in.
502
+ * Measured on the composable gate before this change, at a true risk
503
+ * difference sitting exactly on the production caller's -0.05 margin and a
504
+ * nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.
505
+ * 2. **McNemar's exact test holds a VETO at every non-negative threshold.**
506
+ * Redundant with the interval by construction and kept anyway, so that
507
+ * swapping the estimator for one without that duality cannot silently
508
+ * reintroduce "promotes what the exact test refuses". Witness: n = 6, b = 5,
509
+ * c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs
510
+ * (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE
511
+ * threshold is a noninferiority question, which McNemar's test of "no
512
+ * difference" is not the right test for, so the veto does not apply there.
513
+ * 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it
514
+ * cannot tell a gain from a regression and clears every negative threshold.
515
+ * Away from zero it fails the opposite way: n identical positive deltas give
516
+ * [g, g], which clears threshold 0 on no spread at all. Both are an absence
517
+ * of evidence. Measured on the composable gate before this change, under a
518
+ * bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %
519
+ * false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.
520
+ *
521
+ * Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that
522
+ * picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered
523
+ * exact sign test below it), this picks the ESTIMATOR from the outcome's shape.
524
+ * Both are needed — an exact sign test applied to a tie-pinned median is still
525
+ * blind, and a mean bootstrap CI at n = 6 is still not a valid test.
526
+ */
527
+ /**
528
+ * Which estimator {@link decidePairedPromotion} would use on this data, and the
529
+ * shape facts behind it — for callers that must report the shape on a path
530
+ * where no interval is computed at all (an early rejection, or zero pairs).
531
+ * Cheap: no bootstrap, no interval.
532
+ */
533
+ function pairedDecisionShape(before, after, statistic = "mean") {
534
+ const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after);
535
+ if (statistic === "median") return {
536
+ statistic: "median_bootstrap",
537
+ binaryScale: null,
538
+ tieFraction
539
+ };
540
+ const binaryScale = pairedBinaryScale(before, after);
541
+ if (binaryScale !== null) return {
542
+ statistic: "paired_risk_difference",
543
+ binaryScale,
544
+ tieFraction
545
+ };
546
+ return {
547
+ statistic: "mean_bootstrap",
548
+ binaryScale: null,
549
+ tieFraction
550
+ };
551
+ }
552
+ /**
553
+ * Decide whether a paired candidate-minus-baseline delta clears a promotion
554
+ * threshold. `before` is the baseline arm, `after` the candidate arm, paired by
555
+ * position. Throws on unequal lengths.
556
+ */
557
+ function decidePairedPromotion(before, after, options = {}) {
558
+ if (before.length !== after.length) throw new Error(`decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`);
559
+ const threshold = options.threshold ?? 0;
560
+ if (!Number.isFinite(threshold)) throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`);
561
+ const confidence = options.confidence ?? .95;
562
+ const exactMinimum = minimumPairsForPairedDeltaTest(confidence);
563
+ const requestedMinimum = options.minPairs ?? exactMinimum;
564
+ if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`);
565
+ const minimumPairs = Math.max(requestedMinimum, exactMinimum);
566
+ const n = before.length;
567
+ const sufficient = n >= minimumPairs;
568
+ const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic);
569
+ let core;
570
+ if (binaryScale !== null) {
571
+ const unitControl = before.map((v) => v / binaryScale);
572
+ const unitTreatment = after.map((v) => v / binaryScale);
573
+ const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence);
574
+ const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence);
575
+ const low = score.lower * binaryScale;
576
+ core = {
577
+ statistic: "paired_risk_difference",
578
+ method: "score-interval",
579
+ delta: score.riskDifference * binaryScale,
580
+ low,
581
+ high: score.upper * binaryScale,
582
+ bootstrap: null,
583
+ mcnemar: {
584
+ b: exact.b,
585
+ c: exact.c,
586
+ nDiscordant: exact.nDiscordant,
587
+ pValue: exact.pValue
588
+ },
589
+ pValue: null,
590
+ clearsThreshold: low > threshold,
591
+ label: "success-rate",
592
+ methodDetail: ""
593
+ };
594
+ } else {
595
+ const bootstrapStatistic = options.statistic === "median" ? "median" : "mean";
596
+ const test = pairedDeltaTest(before, after, {
597
+ confidence,
598
+ resamples: options.resamples,
599
+ statistic: bootstrapStatistic,
600
+ seed: options.seed,
601
+ threshold,
602
+ minPairs: options.minPairs
603
+ });
604
+ const ci = test.bootstrap;
605
+ core = {
606
+ statistic: bootstrapStatistic === "mean" ? "mean_bootstrap" : "median_bootstrap",
607
+ method: test.method,
608
+ delta: bootstrapStatistic === "mean" ? ci.mean : ci.median,
609
+ low: ci.low,
610
+ high: ci.high,
611
+ bootstrap: ci,
612
+ mcnemar: null,
613
+ pValue: test.pValue,
614
+ clearsThreshold: test.significant,
615
+ label: bootstrapStatistic,
616
+ methodDetail: test.method === "exact-sign" ? ` Below ${test.minimumPairs} pairs the interval is descriptive only; the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.` : ""
617
+ };
618
+ }
619
+ const indeterminate = !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.low === core.high;
620
+ const indeterminateCause = !indeterminate ? "" : tieFraction === 1 ? "every paired delta is an exact tie" : core.mcnemar !== null && core.mcnemar.nDiscordant === 0 ? "every pair is concordant (0 discordant pairs)" : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`;
621
+ const exactTestVetoes = core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence);
622
+ return {
623
+ n,
624
+ threshold,
625
+ confidence,
626
+ binaryScale,
627
+ tieFraction,
628
+ minimumPairs,
629
+ sufficient,
630
+ indeterminate,
631
+ indeterminateCause,
632
+ exactTestVetoes,
633
+ promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,
634
+ ...core
635
+ };
636
+ }
637
+ function fmt(x) {
638
+ return x.toFixed(4);
639
+ }
640
+ //#endregion
641
+ //#region src/campaign/gates/statistical-heldout.ts
642
+ /**
643
+ * Statistical held-out promotion machinery — the trustworthy core the
644
+ * point-estimate `heldout-delta` gate lacked.
645
+ *
646
+ * The shipped false positive it prevents: a winner re-scored against the
647
+ * baseline on the holdout read run-to-run model NOISE (e.g. 91 vs 95) as a
648
+ * "+4 lift" and shipped, because the gate compared point estimates with no
649
+ * confidence interval. Here we pair candidate vs baseline holdout observations
650
+ * and bootstrap a CI on the paired delta — a candidate ships only when the CI
651
+ * lower bound clears the effect-size threshold (the gain is real at the
652
+ * confidence level, not noise), and is blocked when a critical dimension
653
+ * (e.g. `hallucination_free` for a legal agent) significantly regresses even if
654
+ * the net composite rose (anti-Goodhart).
655
+ *
656
+ * Two traps this module is built around (both produce a NEW false positive if
657
+ * gotten wrong):
658
+ * 1. PAIRING GRANULARITY — pairs by FULL `cellId` (`scenario:rep`), never by
659
+ * `scenarioId` (which averages reps away and destroys the within-pair
660
+ * variance reduction that makes a paired bootstrap tighter than unpaired).
661
+ * One paired observation per cell ⇒ reps multiply n.
662
+ * 2. SCALE — a judge may emit composites/dimensions on [0,1] or 0-100. The
663
+ * threshold + tolerance are interpreted in the judge's NATIVE scale; the
664
+ * per-dimension tolerance auto-scales off the observed baseline magnitudes
665
+ * so `-0.10` on [0,1] doesn't silently become a no-op on a 0-100 dimension.
666
+ */
667
+ /** Tie fraction at/above which a gate annotates its verdict with the tie share.
668
+ * Tie-domination of the median bites structurally at >= 0.5 (the median is then
669
+ * 0 by construction); 0.4 is a softer warn threshold that flags a run APPROACHING
670
+ * that regime, so an operator sees it before the median goes fully blind. */
671
+ const TIE_WARN_FRACTION = .4;
672
+ /**
673
+ * Pair candidate vs baseline holdout observations by FULL cellId. `select`
674
+ * pulls the scalar from a cell's judge reports (composite, or a named
675
+ * dimension); a cell contributes the mean of `select` across its judges. Cells
676
+ * whose scenario is not in `scenarioIds`, or where `select` is undefined for
677
+ * every judge on either side, are skipped on BOTH sides so the arrays stay
678
+ * paired. Throws when the two maps disagree on which holdout cells exist — a
679
+ * load-bearing invariant: the baseline + winner holdout campaigns run the same
680
+ * scenarios with the same seed base, so their cellIds MUST align; a mismatch
681
+ * means a silent pairing bug, not a soft fallback.
682
+ */
683
+ function pairHoldout(candidate, baseline, scenarioIds, select) {
684
+ const cellValue = (byCell, cellId) => {
685
+ const scores = byCell.get(cellId);
686
+ if (!scores) return void 0;
687
+ const vals = [];
688
+ for (const s of Object.values(scores)) {
689
+ if (s.failed === true) throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`);
690
+ const v = select(s);
691
+ if (typeof v === "number" && !Number.isFinite(v)) throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`);
692
+ if (typeof v === "number") vals.push(v);
693
+ }
694
+ if (vals.length === 0) return void 0;
695
+ return vals.reduce((a, b) => a + b, 0) / vals.length;
696
+ };
697
+ const inScope = (cellId) => scenarioIds.has(cellId.split(":")[0] ?? "");
698
+ const candCells = [...candidate.keys()].filter(inScope).sort();
699
+ const baseCells = [...baseline.keys()].filter(inScope).sort();
700
+ if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) throw new Error(`pairHoldout: candidate/baseline holdout cells do not align — candidate=[${candCells.join(",")}] baseline=[${baseCells.join(",")}]. Both holdout campaigns must run the same scenarios with the same seed base.`);
701
+ const before = [];
702
+ const after = [];
703
+ const cellIds = [];
704
+ for (const cellId of candCells) {
705
+ const b = cellValue(baseline, cellId);
706
+ const a = cellValue(candidate, cellId);
707
+ if (b === void 0 && a === void 0) continue;
708
+ if (b === void 0 || a === void 0) throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`);
709
+ before.push(b);
710
+ after.push(a);
711
+ cellIds.push(cellId);
712
+ }
713
+ return {
714
+ before,
715
+ after,
716
+ cellIds
717
+ };
718
+ }
719
+ /**
720
+ * Significance of the held-out composite lift: ship only when the lower bound
721
+ * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
722
+ * 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
723
+ * scale.
724
+ *
725
+ * The decision is delegated whole to {@link decidePairedPromotion}, the one
726
+ * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
727
+ * also calls. That module's header carries the measurements; the short version
728
+ * is three guards a bare `bootstrap.low > threshold` does not have:
729
+ *
730
+ * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
731
+ * only paired-binary construction that stays valid at a nonzero margin;
732
+ * - McNemar's exact test VETOES at any non-negative threshold;
733
+ * - a ZERO-WIDTH interval is refused rather than promoted, in either
734
+ * direction — [0,0] clears every negative threshold and [g,g] clears every
735
+ * threshold below g, and both are an absence of evidence, not a result.
736
+ *
737
+ * Measured on this function before those guards landed, at a nominal 5 %:
738
+ * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
739
+ * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
740
+ * delta is exactly 0.
741
+ *
742
+ * At small n, where the percentile bootstrap is descriptive only, a
743
+ * pre-registered exact sign test still carries the bootstrap path.
744
+ */
745
+ function heldoutSignificance(paired, opts = {}) {
746
+ const deltaThreshold = opts.deltaThreshold ?? 0;
747
+ const confidence = opts.confidence ?? .95;
748
+ const resamples = opts.resamples ?? 2e3;
749
+ const seed = opts.seed ?? 1337;
750
+ const statistic = opts.statistic ?? "mean";
751
+ const decision = decidePairedPromotion(paired.before, paired.after, {
752
+ confidence,
753
+ resamples,
754
+ statistic,
755
+ seed,
756
+ threshold: deltaThreshold,
757
+ minPairs: opts.minProductiveRuns
758
+ });
759
+ const bootstrap = decision.bootstrap ?? pairedBootstrap(paired.before, paired.after, {
760
+ confidence,
761
+ resamples,
762
+ statistic,
763
+ seed
764
+ });
765
+ const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
766
+ confidence,
767
+ resamples,
768
+ statistic: "median",
769
+ seed
770
+ });
771
+ const n = paired.before.length;
772
+ let ties = 0;
773
+ for (let i = 0; i < n; i += 1) {
774
+ const after = paired.after[i] ?? 0;
775
+ const before = paired.before[i] ?? 0;
776
+ if (Math.abs(after - before) < 1e-9) ties += 1;
777
+ }
778
+ const tieFraction = n === 0 ? 0 : ties / n;
779
+ return {
780
+ paired,
781
+ bootstrap,
782
+ medianBootstrap,
783
+ decision,
784
+ decisionStatistic: decision.statistic,
785
+ mcnemar: decision.mcnemar,
786
+ tieFraction,
787
+ n,
788
+ minimumRequired: decision.minimumPairs,
789
+ decisionMethod: decision.method,
790
+ pValue: decision.pValue,
791
+ significant: decision.promote,
792
+ fewRuns: !decision.sufficient
793
+ };
794
+ }
795
+ /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
796
+ * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
797
+ * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
798
+ function detectScale(values) {
799
+ return values.some((v) => Math.abs(v) > 1.5) ? 100 : 1;
800
+ }
801
+ /** Per-critical-dimension regression guard. For each dimension, pair the
802
+ * candidate vs baseline values by full cellId and bootstrap the paired delta;
803
+ * a dimension is "regressed" when the CI lower bound < −tolerance (conservative
804
+ * — blocks if the credible worst case exceeds tolerance, which is the right
805
+ * posture for safety dimensions like `hallucination_free`). When `tolerance`
806
+ * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
807
+ *
808
+ * The interval comes from {@link decidePairedPromotion}, so a pass/fail
809
+ * dimension is judged on Tango's score interval rather than a percentile
810
+ * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
811
+ * is not a valid interval at one. That matters most here because this guard
812
+ * fails OPEN by construction: `tolerance` is positive, so an interval pinned at
813
+ * [0,0] never satisfies `low < −tolerance` and a real regression on a safety
814
+ * dimension would be reported as `regressed: false`. On the median it fails the
815
+ * same way for the same reason — when most pairs tie, which is automatic for a
816
+ * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
817
+ * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
818
+ * restore the pre-0.134 behaviour. */
819
+ function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
820
+ const out = [];
821
+ for (const dim of criticalDimensions) {
822
+ const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
823
+ if (paired.before.length === 0) continue;
824
+ const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
825
+ const bootstrapStatistic = opts.statistic ?? "mean";
826
+ const shared = {
827
+ confidence: opts.confidence ?? .95,
828
+ resamples: opts.resamples ?? 2e3,
829
+ statistic: bootstrapStatistic,
830
+ seed: opts.seed ?? 1337
831
+ };
832
+ const guard = decidePairedPromotion(paired.before, paired.after, shared);
833
+ const regression = decidePairedPromotion(paired.after, paired.before, {
834
+ ...shared,
835
+ threshold: tolerance
836
+ });
837
+ const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared);
838
+ out.push({
839
+ dimension: dim,
840
+ bootstrap,
841
+ bootstrapStatistic,
842
+ ci: {
843
+ low: guard.low,
844
+ high: guard.high
845
+ },
846
+ decisionStatistic: guard.statistic,
847
+ mcnemar: guard.mcnemar,
848
+ indeterminate: guard.indeterminate,
849
+ regressed: bootstrap.low < -tolerance || regression.promote,
850
+ tolerance,
851
+ n: paired.before.length
852
+ });
853
+ }
854
+ return out;
855
+ }
856
+ //#endregion
857
+ //#region src/campaign/gates/power-preflight.ts
858
+ /** Two-sided z for the common confidence levels; interpolation is overkill here. */
859
+ function zFor(confidence) {
860
+ if (confidence >= .99) return 2.576;
861
+ if (confidence >= .95) return 1.96;
862
+ if (confidence >= .9) return 1.645;
863
+ return 1.282;
864
+ }
865
+ /** Estimate the minimum detectable lift a paired-holdout improvement run can
866
+ * ship at a given budget, from the baseline holdout composites — call it BEFORE
867
+ * spending a search to learn whether the effect you are hunting is even
868
+ * observable at this holdout size and worker variance. */
869
+ function powerPreflight(opts) {
870
+ const composites = opts.baselineComposites.filter((v) => Number.isFinite(v));
871
+ if (composites.length < 3) throw new Error(`powerPreflight: need >= 3 finite baseline composites to estimate variance, got ${composites.length}`);
872
+ const deltaThreshold = opts.deltaThreshold ?? .05;
873
+ const confidence = opts.confidence ?? .95;
874
+ const n = opts.pairedN ?? composites.length;
875
+ if (n < 2) throw new Error(`powerPreflight: pairedN must be >= 2, got ${n}`);
876
+ const mean = composites.reduce((a, b) => a + b, 0) / composites.length;
877
+ const variance = composites.reduce((a, b) => a + (b - mean) * (b - mean), 0) / (composites.length - 1);
878
+ const sd = Math.sqrt(variance);
879
+ const z = zFor(confidence);
880
+ const mde = deltaThreshold + z * Math.SQRT2 * sd / Math.sqrt(n);
881
+ const scaleAssumed = composites.every((v) => v >= -.001 && v <= 1.5);
882
+ const headroom = Math.max(0, 1 - mean);
883
+ const underpowered = scaleAssumed && mde > headroom;
884
+ const sharedChannelCaveat = opts.sharedScorerChannel ? "Holdout and gate share one scoring channel: raising n/reps reduces only idiosyncratic noise — systematic judge bias remains and this MDE is a lower bound. Full debiasing needs an independent second scoring channel (different judge/benchmark family)." : void 0;
885
+ const recommendation = underpowered ? `UNDERPOWERED: minimum detectable lift ${mde.toFixed(3)} exceeds the ${headroom.toFixed(3)} headroom above the baseline (${mean.toFixed(3)}) — no achievable effect can ship at this budget. Raise paired n (scenarios x reps) to ~${Math.ceil((z * Math.SQRT2 * sd / Math.max(headroom - deltaThreshold, .01)) ** 2)} or reduce worker variance before searching.` : `Minimum detectable lift at n=${n}: ${mde.toFixed(3)} (baseline sd ${sd.toFixed(3)}). Effects smaller than this cannot clear the gate; budget the search for effects you believe exceed it.`;
886
+ return {
887
+ n,
888
+ sd,
889
+ mde,
890
+ baselineMean: mean,
891
+ headroom,
892
+ underpowered,
893
+ scaleAssumed,
894
+ deltaThreshold,
895
+ confidence,
896
+ ...sharedChannelCaveat ? { sharedChannelCaveat } : {},
897
+ recommendation: sharedChannelCaveat ? `${recommendation} ${sharedChannelCaveat}` : recommendation
898
+ };
899
+ }
900
+ //#endregion
901
+ //#region src/json-recovery.ts
902
+ /**
903
+ * Truncation-tolerant JSON recovery — shared by every parser that reads JSON
904
+ * out of a model response (reflective-mutation proposals, judge scores, the
905
+ * completion-correctness checker).
906
+ *
907
+ * LLMs routinely hit a max_tokens cap mid-emission, leaving a JSON prefix
908
+ * with an unclosed string / object / array and often a dangling key or
909
+ * trailing comma. Throwing on that prefix — and letting the throw fold into
910
+ * a fabricated zero score downstream — is the bug class this module exists
911
+ * to prevent (see `JudgeParseError`'s contract: a synthetic zero is
912
+ * indistinguishable from a real low score). Recovering the complete prefix
913
+ * turns a would-be fabricated zero into a real measurement.
914
+ */
915
+ /**
916
+ * Walk the input as JSON-aware (string vs not, escape-aware) and close
917
+ * unclosed `{` / `[` in LIFO order at the tail. If the input was already
918
+ * balanced returns it unchanged. If a string was open at end-of-input we
919
+ * also close it with `"` first, since a truncated string-mid-value is the
920
+ * most common LLM cap-hit failure mode and JSON.parse cannot proceed
921
+ * without one.
922
+ *
923
+ * Returns null when the structure is unrecoverable (e.g. depth would go
924
+ * negative — that's an *over*-closed prefix, not a truncation).
925
+ */
926
+ function autoCloseTruncatedJson(raw) {
927
+ const stack = [];
928
+ let inString = false;
929
+ let escaped = false;
930
+ for (const c of raw) {
931
+ if (escaped) {
932
+ escaped = false;
933
+ continue;
934
+ }
935
+ if (inString) {
936
+ if (c === "\\") {
937
+ escaped = true;
938
+ continue;
939
+ }
940
+ if (c === "\"") {
941
+ inString = false;
942
+ continue;
943
+ }
944
+ continue;
945
+ }
946
+ if (c === "\"") {
947
+ inString = true;
948
+ continue;
949
+ }
950
+ if (c === "{" || c === "[") stack.push(c);
951
+ else if (c === "}") {
952
+ if (stack.pop() !== "{") return null;
953
+ } else if (c === "]") {
954
+ if (stack.pop() !== "[") return null;
955
+ }
956
+ }
957
+ if (stack.length === 0 && !inString) return raw;
958
+ let suffix = "";
959
+ if (escaped) suffix += "\\";
960
+ if (inString) suffix += "\"";
961
+ while (stack.length > 0) {
962
+ const opener = stack.pop();
963
+ suffix += opener === "{" ? "}" : "]";
964
+ }
965
+ return raw + suffix;
966
+ }
967
+ const UNPARSEABLE = Symbol("unparseable");
968
+ function tryParse(candidate) {
969
+ try {
970
+ return JSON.parse(candidate);
971
+ } catch {
972
+ return UNPARSEABLE;
973
+ }
974
+ }
975
+ /** Index of the last `,` that sits outside any string literal, or -1. Cutting
976
+ * there discards a dangling key / half-emitted value at the tail while
977
+ * keeping every complete member before it. */
978
+ function lastCommaOutsideString(s) {
979
+ let inString = false;
980
+ let escaped = false;
981
+ let last = -1;
982
+ for (let i = 0; i < s.length; i++) {
983
+ const c = s[i];
984
+ if (escaped) {
985
+ escaped = false;
986
+ continue;
987
+ }
988
+ if (inString) {
989
+ if (c === "\\") escaped = true;
990
+ else if (c === "\"") inString = false;
991
+ continue;
992
+ }
993
+ if (c === "\"") inString = true;
994
+ else if (c === ",") last = i;
995
+ }
996
+ return last;
997
+ }
998
+ /**
999
+ * Best-effort parse of a possibly-truncated JSON payload embedded in model
1000
+ * output (prose and markdown fences tolerated). Tries each opener position
1001
+ * (earliest `{`/`[` first, then the other when the first yields nothing —
1002
+ * prose like `[note] {"score":3}` must not poison the slice), and from each
1003
+ * start, in order:
1004
+ *
1005
+ * 1. plain `JSON.parse` of the opener → last-closer slice,
1006
+ * 2. auto-closing unclosed structures at the tail
1007
+ * (`autoCloseTruncatedJson`),
1008
+ * 3. trimming the tail back to the previous complete member boundary (the
1009
+ * last comma outside a string) and auto-closing again, repeatedly.
1010
+ *
1011
+ * Recovers e.g. `{"correct": false, "` → `{ correct: false }` — the exact
1012
+ * cap-hit shape that has zeroed real eval rows. Returns the parsed value
1013
+ * (always an object or array, given the slice starts at an opener), or
1014
+ * `null` when nothing parseable can be recovered. Never throws.
1015
+ */
1016
+ function recoverTruncatedJson(text) {
1017
+ const starts = [text.indexOf("{"), text.indexOf("[")].filter((i) => i >= 0).sort((a, b) => a - b);
1018
+ for (const start of starts) {
1019
+ const recovered = recoverFrom(text.slice(start));
1020
+ if (recovered !== UNPARSEABLE) return recovered;
1021
+ }
1022
+ return null;
1023
+ }
1024
+ function recoverFrom(slice) {
1025
+ let candidate = slice;
1026
+ const lastClose = Math.max(candidate.lastIndexOf("}"), candidate.lastIndexOf("]"));
1027
+ if (lastClose > 0) {
1028
+ const balanced = tryParse(candidate.slice(0, lastClose + 1));
1029
+ if (balanced !== UNPARSEABLE) return balanced;
1030
+ }
1031
+ for (let i = 0; i < 64; i++) {
1032
+ const closed = autoCloseTruncatedJson(candidate);
1033
+ if (closed !== null) {
1034
+ const parsed = tryParse(closed);
1035
+ if (parsed !== UNPARSEABLE) return parsed;
1036
+ }
1037
+ const cut = lastCommaOutsideString(candidate);
1038
+ if (cut <= 0) return UNPARSEABLE;
1039
+ candidate = candidate.slice(0, cut);
1040
+ }
1041
+ return UNPARSEABLE;
1042
+ }
1043
+ //#endregion
1044
+ //#region src/reflective-mutation.ts
1045
+ /**
1046
+ * Reflective mutation — primitives for trace-conditioned prompt rewriting.
1047
+ *
1048
+ * Used by `prompt-evolution.ts` (and any consumer running iterative
1049
+ * improvement). Given a parent prompt + concrete trace evidence (top trials,
1050
+ * bottom trials, missed expectations), produce an LLM-ready prompt that
1051
+ * proposes targeted mutations — not blind rephrasings.
1052
+ *
1053
+ * Why this lives outside `prompt-evolution.ts`: any consumer that wants to
1054
+ * run reflective rewriting WITHOUT the population/Pareto machinery can
1055
+ * import these primitives directly.
1056
+ *
1057
+ * Quality bar (vs. naive "mutate this prompt"):
1058
+ * - Show parent ↔ children diff, not just one variant
1059
+ * - Quote specific missed goldens with their match phrases
1060
+ * - Surface the model's actual emitted output side-by-side with what was expected
1061
+ * - Quote concrete mutation primitives so the model has a vocabulary
1062
+ */
1063
+ /** Bound on rendered/carried `emitted` evidence. ONE constant shared by the
1064
+ * producer (campaignBreakdown's per-scenario excerpt) and this renderer — if
1065
+ * the two drifted, the tighter side would silently re-clip carried evidence. */
1066
+ const EMITTED_EVIDENCE_MAX_CHARS = 2e3;
1067
+ const DEFAULT_MUTATION_PRIMITIVES = [
1068
+ "Strengthen an imperative (\"should\" → \"must\")",
1069
+ "Add a concrete example pulled from a missed-golden phrase",
1070
+ "Remove a redundant rule that did not improve recall",
1071
+ "Add a counterfactual (\"if X is missing, the score is capped at Y\")",
1072
+ "Reorder sections so the highest-impact rule is first",
1073
+ "Replace abstract language with a domain-specific noun the trial misses"
1074
+ ];
1075
+ /**
1076
+ * Build the LLM-ready reflection prompt. Output is plain text — pass it as
1077
+ * the user message. The system message should be small and stable (e.g.
1078
+ * "Output ONLY a JSON object matching the schema below.").
1079
+ */
1080
+ function buildReflectionPrompt(ctx) {
1081
+ const primitives = ctx.mutationPrimitives ?? DEFAULT_MUTATION_PRIMITIVES;
1082
+ const sections = [];
1083
+ sections.push(`# Mutation target: ${ctx.target}`);
1084
+ sections.push("");
1085
+ sections.push(`You are tuning the prompt component named \`${ctx.target}\`. The current variant is shown below; you have ${ctx.topTrials.length} top trials and ${ctx.bottomTrials.length} bottom trials as evidence. Propose ${ctx.childCount} mutation${ctx.childCount === 1 ? "" : "s"} that fix specific weaknesses visible in the bottom trials. Avoid blank rephrasings.`);
1086
+ sections.push("");
1087
+ sections.push("## Current variant");
1088
+ sections.push("```json");
1089
+ sections.push(JSON.stringify(ctx.parentPayload, null, 2));
1090
+ sections.push("```");
1091
+ sections.push("");
1092
+ if (ctx.bottomTrials.length > 0) {
1093
+ sections.push("## Failures (bottom trials) — what went wrong");
1094
+ sections.push("");
1095
+ for (const trial of ctx.bottomTrials) {
1096
+ sections.push(`### Trial \`${trial.id}\` — score ${trial.score.toFixed(2)}${trial.inputName ? ` (${trial.inputName})` : ""}`);
1097
+ if (trial.failureNote) {
1098
+ sections.push("");
1099
+ sections.push(`**Why it scored low:** ${truncate(trial.failureNote, 1500)}`);
1100
+ }
1101
+ const missed = (trial.expectations ?? []).filter((e) => !e.matched);
1102
+ if (missed.length > 0) {
1103
+ sections.push("");
1104
+ sections.push("**Missed expectations:**");
1105
+ for (const m of missed) sections.push(`- \`${m.id}\`: should match phrase \`${quote(m.phrase)}\``);
1106
+ }
1107
+ if (trial.emitted) {
1108
+ sections.push("");
1109
+ sections.push("**What the agent emitted:**");
1110
+ sections.push("```");
1111
+ sections.push(truncate(trial.emitted, EMITTED_EVIDENCE_MAX_CHARS));
1112
+ sections.push("```");
1113
+ }
1114
+ sections.push("");
1115
+ }
1116
+ }
1117
+ if (ctx.topTrials.length > 0) {
1118
+ sections.push("## Successes (top trials) — what to preserve");
1119
+ sections.push("");
1120
+ for (const trial of ctx.topTrials) sections.push(`- \`${trial.id}\`: score ${trial.score.toFixed(2)}${trial.inputName ? ` (${trial.inputName})` : ""}`);
1121
+ sections.push("");
1122
+ }
1123
+ sections.push("## Allowed mutation primitives");
1124
+ sections.push("");
1125
+ for (const p of primitives) sections.push(`- ${p}`);
1126
+ sections.push("");
1127
+ sections.push("## Output schema");
1128
+ sections.push("");
1129
+ sections.push("Respond with a JSON object — no prose, no markdown fences:");
1130
+ sections.push("```json");
1131
+ sections.push(JSON.stringify({ proposals: [{
1132
+ label: "<short label, ≤ 40 chars>",
1133
+ rationale: "<which failure this targets and which primitive you used>",
1134
+ payload: "<full payload of the new variant — same shape as the current variant>"
1135
+ }] }, null, 2));
1136
+ sections.push("```");
1137
+ return sections.join("\n");
1138
+ }
1139
+ function truncate(s, max) {
1140
+ if (s.length <= max) return s;
1141
+ return `${s.slice(0, max)}… [truncated]`;
1142
+ }
1143
+ function quote(s) {
1144
+ return s.replace(/`/g, "\\`");
1145
+ }
1146
+ /**
1147
+ * Parse the model's JSON response back into proposals. Tolerates markdown
1148
+ * fences and surrounding prose. Returns at most `maxProposals`.
1149
+ */
1150
+ function parseReflectionResponse(raw, maxProposals) {
1151
+ let text = raw.trim();
1152
+ if (text.startsWith("```")) text = text.replace(/^```(?:json)?\n?/, "").replace(/\n?```$/, "");
1153
+ let parsed = null;
1154
+ const objectStart = text.indexOf("{");
1155
+ const objectEnd = text.lastIndexOf("}");
1156
+ const arrayStart = text.indexOf("[");
1157
+ const arrayEnd = text.lastIndexOf("]");
1158
+ const tryObjectFirst = objectStart >= 0 && (arrayStart < 0 || objectStart < arrayStart);
1159
+ const candidates = [];
1160
+ if (tryObjectFirst) {
1161
+ if (objectStart >= 0 && objectEnd > objectStart) candidates.push(text.slice(objectStart, objectEnd + 1));
1162
+ if (arrayStart >= 0 && arrayEnd > arrayStart) candidates.push(text.slice(arrayStart, arrayEnd + 1));
1163
+ } else {
1164
+ if (arrayStart >= 0 && arrayEnd > arrayStart) candidates.push(text.slice(arrayStart, arrayEnd + 1));
1165
+ if (objectStart >= 0 && objectEnd > objectStart) candidates.push(text.slice(objectStart, objectEnd + 1));
1166
+ }
1167
+ for (const slice of candidates) try {
1168
+ parsed = JSON.parse(slice);
1169
+ break;
1170
+ } catch {}
1171
+ if (parsed == null) for (const slice of candidates) {
1172
+ const closed = autoCloseTruncatedJson(slice);
1173
+ if (closed != null && closed !== slice) try {
1174
+ parsed = JSON.parse(closed);
1175
+ break;
1176
+ } catch {}
1177
+ }
1178
+ if (parsed == null) return [];
1179
+ let proposalsRaw;
1180
+ if (Array.isArray(parsed)) proposalsRaw = parsed;
1181
+ else if (parsed && typeof parsed === "object") proposalsRaw = parsed.proposals;
1182
+ if (!Array.isArray(proposalsRaw)) return [];
1183
+ const out = [];
1184
+ for (const p of proposalsRaw) {
1185
+ if (!p || typeof p !== "object") continue;
1186
+ const obj = p;
1187
+ if (!("payload" in obj)) continue;
1188
+ out.push({
1189
+ label: typeof obj.label === "string" ? obj.label : "mutation",
1190
+ rationale: typeof obj.rationale === "string" ? obj.rationale : "",
1191
+ payload: obj.payload
1192
+ });
1193
+ if (maxProposals !== void 0 && out.length >= maxProposals) break;
1194
+ }
1195
+ return out;
1196
+ }
1197
+ //#endregion
1198
+ //#region src/campaign/score-utils.ts
1199
+ /**
1200
+ * Shared campaign-score reductions used by every optimizer preset
1201
+ * (`runOptimization`, external optimization methods, `compareOptimizationMethods`).
1202
+ * "composite of a campaign" and "per-scenario / per-dimension breakdown" so
1203
+ * the optimizers cannot drift on how a surface's score is computed.
1204
+ */
1205
+ /** Mean composite across cells with complete task-quality evidence.
1206
+ * Partial judge results remain on their cells but never enter this value.
1207
+ * A campaign with no complete score has no numeric mean and fails loudly. */
1208
+ function campaignMeanComposite(campaign) {
1209
+ const mean = campaignMeanCompositeOrNull(campaign);
1210
+ if (mean === null) throw new Error("campaignMeanComposite: campaign has no complete cell-quality scores");
1211
+ return mean;
1212
+ }
1213
+ /** Nullable campaign mean for wire and report fields that represent missing quality. */
1214
+ function campaignMeanCompositeOrNull(campaign) {
1215
+ const scores = campaign.cells.flatMap((cell) => {
1216
+ const score = projectCampaignCellQuality(cell).score;
1217
+ return score === void 0 ? [] : [score];
1218
+ });
1219
+ return scores.length === 0 ? null : scores.reduce((sum, score) => sum + score, 0) / scores.length;
1220
+ }
1221
+ /** Reject rank keys that cannot produce deterministic lexicographic ordering. */
1222
+ function assertFiniteRankKey(key, label, expectedLength) {
1223
+ if (!Array.isArray(key) || key.length === 0) throw new Error(`${label} must return a non-empty array`);
1224
+ if (expectedLength !== void 0 && key.length !== expectedLength) throw new Error(`${label} returned ${key.length} elements; expected ${expectedLength}`);
1225
+ for (let index = 0; index < key.length; index++) if (!Number.isFinite(key[index])) throw new Error(`${label}[${index}] must be finite`);
1226
+ }
1227
+ /** Compare fixed-length lexicographic rank keys where each element is higher-is-better.
1228
+ * Returns a positive number when `a` ranks above `b`, negative when below, and
1229
+ * zero when equal. */
1230
+ function compareRankKeys(a, b) {
1231
+ assertFiniteRankKey(a, "rank key a");
1232
+ assertFiniteRankKey(b, "rank key b", a.length);
1233
+ for (let i = 0; i < a.length; i++) {
1234
+ const av = a[i];
1235
+ const bv = b[i];
1236
+ if (av !== bv) return av - bv;
1237
+ }
1238
+ return 0;
1239
+ }
1240
+ /** Per-candidate evidence a reflective/patch proposer grounds its next proposal
1241
+ * on: mean score per judge dimension + per-scenario composite. */
1242
+ function campaignBreakdown(campaign) {
1243
+ const dimSums = {};
1244
+ const dimCounts = {};
1245
+ const byScenario = /* @__PURE__ */ new Map();
1246
+ const notesByScenario = /* @__PURE__ */ new Map();
1247
+ const emittedByScenario = /* @__PURE__ */ new Map();
1248
+ for (const cell of campaign.cells) {
1249
+ const quality = projectCampaignCellQuality(cell);
1250
+ if (quality.score === void 0) continue;
1251
+ const judgeScores = Object.values(quality.successfulJudgeScores);
1252
+ const cellComposite = quality.score;
1253
+ const arr = byScenario.get(cell.scenarioId) ?? [];
1254
+ arr.push(cellComposite);
1255
+ byScenario.set(cell.scenarioId, arr);
1256
+ if (typeof cell.artifact === "string" && cell.artifact.trim().length > 0) {
1257
+ const prev = emittedByScenario.get(cell.scenarioId);
1258
+ if (!prev || cellComposite < prev.composite) emittedByScenario.set(cell.scenarioId, {
1259
+ composite: cellComposite,
1260
+ text: cell.artifact.slice(0, EMITTED_EVIDENCE_MAX_CHARS)
1261
+ });
1262
+ }
1263
+ for (const s of judgeScores) if (s.notes?.trim()) {
1264
+ const set = notesByScenario.get(cell.scenarioId) ?? /* @__PURE__ */ new Set();
1265
+ set.add(s.notes.trim());
1266
+ notesByScenario.set(cell.scenarioId, set);
1267
+ }
1268
+ for (const score of judgeScores) for (const [key, value] of Object.entries(score.dimensions)) {
1269
+ if (!Number.isFinite(value)) continue;
1270
+ dimSums[key] = (dimSums[key] ?? 0) + value;
1271
+ dimCounts[key] = (dimCounts[key] ?? 0) + 1;
1272
+ }
1273
+ }
1274
+ const dimensions = {};
1275
+ for (const key of Object.keys(dimSums)) {
1276
+ const count = dimCounts[key] ?? 0;
1277
+ dimensions[key] = count > 0 ? (dimSums[key] ?? 0) / count : 0;
1278
+ }
1279
+ return {
1280
+ dimensions,
1281
+ scenarios: [...byScenario.entries()].map(([scenarioId, comps]) => {
1282
+ const notesSet = notesByScenario.get(scenarioId);
1283
+ const notes = notesSet && notesSet.size > 0 ? [...notesSet].join(" | ") : void 0;
1284
+ const emitted = emittedByScenario.get(scenarioId)?.text;
1285
+ return {
1286
+ scenarioId,
1287
+ composite: comps.reduce((a, b) => a + b, 0) / comps.length,
1288
+ ...notes ? { notes } : {},
1289
+ ...emitted ? { emitted } : {}
1290
+ };
1291
+ })
1292
+ };
1293
+ }
1294
+ //#endregion
1295
+ //#region src/campaign/provenance.ts
1296
+ /**
1297
+ * Loop provenance — the durable, queryable record of WHAT a self-improvement
1298
+ * loop did and WHY, plus the OTel spans that let an OTLP collector pivot from
1299
+ * an eval-run to the underlying candidate→cell→gate→promote chain.
1300
+ *
1301
+ * Two artifacts, one source of truth:
1302
+ *
1303
+ * 1. `LoopProvenanceRecord` — a structured JSON record capturing every
1304
+ * candidate (surfaceHash + label + rationale + structured cause), its measured composite,
1305
+ * the gate decision + reasons + delta, the held-out lift, the explicit
1306
+ * baseline→candidate diff, and BACKEND PROVENANCE (the
1307
+ * `assertRealBackend` verdict + worker call count + model). This is the
1308
+ * ingestable audit artifact: the +lift recomputes from it, the "because
1309
+ * Z" rationale survives in it, and a stub backend is detectable from it.
1310
+ *
1311
+ * 2. `loopProvenanceSpans()` — the same chain emitted as OTLP-ingestable
1312
+ * `TraceSpanEvent`s, pivoted on the substrate's standard
1313
+ * `tangle.runId` / `tangle.scenarioId` / `tangle.cellId` /
1314
+ * `tangle.generation` attributes (the same pivots `/adapters/otel`
1315
+ * reads). The hosted `/v1/ingest/traces` endpoint receives the FULL loop,
1316
+ * not just the `cost.*` spans `runCampaign` already emits per cell.
1317
+ *
1318
+ * The record is built from the loop result and its settled cost receipts — no
1319
+ * second usage collector can contradict what the measured cells recorded.
1320
+ */
1321
+ /** One translation from a completed improvement loop into durable evidence. */
1322
+ function loopProvenanceArgsFromResult(input) {
1323
+ const { result } = input;
1324
+ return {
1325
+ runId: input.runId,
1326
+ runDir: input.runDir,
1327
+ timestamp: input.timestamp,
1328
+ baselineSurface: input.baselineSurface,
1329
+ winnerSurface: result.winnerSurface,
1330
+ ...result.winnerLabel ? { winnerLabel: result.winnerLabel } : {},
1331
+ ...result.winnerRationale ? { winnerRationale: result.winnerRationale } : {},
1332
+ baselineSearchCampaign: result.baselineCampaign,
1333
+ generations: result.generations.map(({ record, surfaces }) => ({
1334
+ generationIndex: record.generationIndex,
1335
+ candidates: record.candidates,
1336
+ promoted: record.promoted,
1337
+ surfaces: surfaces.map(({ surfaceHash, surface, campaign }) => ({
1338
+ surfaceHash,
1339
+ surface,
1340
+ campaign
1341
+ }))
1342
+ })),
1343
+ gate: result.gateResult,
1344
+ ...result.holdout === "deferred" ? { holdout: "deferred" } : {},
1345
+ baselineOnHoldout: result.baselineOnHoldout,
1346
+ winnerOnHoldout: result.winnerOnHoldout,
1347
+ ...result.neutralizedSurface && result.neutralizedOnHoldout ? {
1348
+ neutralizedSurface: result.neutralizedSurface,
1349
+ neutralizedOnHoldout: result.neutralizedOnHoldout
1350
+ } : {},
1351
+ costReceipts: input.costReceipts,
1352
+ totalCostUsd: input.totalCostUsd,
1353
+ totalDurationMs: input.totalDurationMs
1354
+ };
1355
+ }
1356
+ function meanHoldoutComposite(campaign) {
1357
+ return campaignMeanComposite(campaign);
1358
+ }
1359
+ /** Build the durable provenance record from a completed loop result. */
1360
+ function buildLoopProvenanceRecord(args) {
1361
+ if (!args.runId.trim() || !args.runDir.trim()) throw new Error("buildLoopProvenanceRecord: runId and runDir must be non-empty");
1362
+ const timestampMs = Date.parse(args.timestamp);
1363
+ if (!Number.isFinite(timestampMs) || new Date(timestampMs).toISOString() !== args.timestamp) throw new Error("buildLoopProvenanceRecord: timestamp must be a canonical ISO instant");
1364
+ assertGateContributions(args.gate.contributingGates, "buildLoopProvenanceRecord");
1365
+ const agentReceipts = args.costReceipts.filter((receipt) => receipt.channel === "agent");
1366
+ const integrity = summarizeAgentReceiptIntegrity(agentReceipts);
1367
+ const models = [...new Set(agentReceipts.map((receipt) => receipt.model))].sort();
1368
+ const baselineSearchComposite = campaignMeanComposite(args.baselineSearchCampaign);
1369
+ if (!Number.isFinite(baselineSearchComposite)) throw new Error("buildLoopProvenanceRecord: baselineSearchComposite must be finite");
1370
+ const candidates = [];
1371
+ let incumbentSurfaceHash = surfaceHash(args.baselineSurface);
1372
+ let incumbentComposite = baselineSearchComposite;
1373
+ let previousGeneration = -1;
1374
+ for (const gen of args.generations) {
1375
+ if (!Number.isSafeInteger(gen.generationIndex) || gen.generationIndex !== previousGeneration + 1) throw new Error("buildLoopProvenanceRecord: generation indices must be contiguous integers starting at zero");
1376
+ previousGeneration = gen.generationIndex;
1377
+ if (gen.candidates.length === 0) throw new Error("buildLoopProvenanceRecord: a recorded generation must contain a candidate");
1378
+ if (new Set(gen.promoted).size !== gen.promoted.length || gen.promoted.length > 1) throw new Error("buildLoopProvenanceRecord: each generation may promote at most one candidate");
1379
+ const promotedSet = new Set(gen.promoted);
1380
+ const surfaceByHash = new Map(gen.surfaces.map((measured) => [measured.surfaceHash, measured]));
1381
+ const candidateByHash = new Map(gen.candidates.map((candidate) => [candidate.surfaceHash, candidate]));
1382
+ if (candidateByHash.size !== gen.candidates.length) throw new Error("buildLoopProvenanceRecord: duplicate candidate surface hash");
1383
+ if (surfaceByHash.size !== gen.surfaces.length) throw new Error("buildLoopProvenanceRecord: duplicate candidate surface entry");
1384
+ if (surfaceByHash.size !== candidateByHash.size) throw new Error("buildLoopProvenanceRecord: every measured candidate requires exactly one surface");
1385
+ for (const promotedHash of promotedSet) if (!candidateByHash.has(promotedHash)) throw new Error("buildLoopProvenanceRecord: promoted hash has no measured candidate");
1386
+ for (const c of gen.candidates) {
1387
+ validateCandidateMeasurement(c, incumbentSurfaceHash, incumbentComposite, promotedSet.has(c.surfaceHash));
1388
+ const measured = surfaceByHash.get(c.surfaceHash);
1389
+ if (measured === void 0) throw new Error("buildLoopProvenanceRecord: measured candidate is missing its surface");
1390
+ const { surface, campaign } = measured;
1391
+ if (!surfaceHashMatches(surface, c.surfaceHash)) throw new Error("buildLoopProvenanceRecord: candidate surface hash does not match its surface bytes");
1392
+ if (campaign.splitDigest !== args.baselineSearchCampaign.splitDigest) throw new Error("buildLoopProvenanceRecord: candidate campaign does not match the search split");
1393
+ const entry = {
1394
+ generation: gen.generationIndex,
1395
+ surfaceHash: c.surfaceHash,
1396
+ contentHash: surfaceContentHash(surface),
1397
+ campaignDigest: campaignMeasurementDigest(campaign),
1398
+ parentSurfaceHash: c.parentSurfaceHash,
1399
+ parentComposite: c.parentComposite,
1400
+ eligibleForPromotion: c.eligibleForPromotion,
1401
+ coverage: {
1402
+ expectedCells: c.coverage.expectedCells,
1403
+ scorableCells: c.coverage.scorableCells,
1404
+ unscorableCells: c.coverage.unscorableCells.map((cell) => ({ ...cell }))
1405
+ },
1406
+ composite: c.composite,
1407
+ promoted: promotedSet.has(c.surfaceHash)
1408
+ };
1409
+ if (c.label) entry.label = c.label;
1410
+ if (c.rationale) entry.rationale = c.rationale;
1411
+ if (c.attribution) entry.attribution = c.attribution;
1412
+ if (c.observedDeltaFromParent !== void 0) entry.observedDeltaFromParent = c.observedDeltaFromParent;
1413
+ candidates.push(entry);
1414
+ }
1415
+ const promotedHash = gen.promoted[0];
1416
+ if (promotedHash) {
1417
+ const promoted = candidateByHash.get(promotedHash);
1418
+ incumbentSurfaceHash = promoted.surfaceHash;
1419
+ if (promoted.composite === null) throw new Error("buildLoopProvenanceRecord: promoted candidate is missing a composite");
1420
+ incumbentComposite = promoted.composite;
1421
+ }
1422
+ }
1423
+ if (surfaceHash(args.winnerSurface) !== incumbentSurfaceHash) throw new Error("buildLoopProvenanceRecord: winner surface does not match the final promoted incumbent");
1424
+ const holdoutDeferred = args.holdout === "deferred";
1425
+ if (args.baselineOnHoldout.splitDigest !== args.winnerOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: baseline and winner use different holdout splits");
1426
+ if (args.neutralizedSurface === void 0 !== (args.neutralizedOnHoldout === void 0)) throw new Error("buildLoopProvenanceRecord: neutralized surface and campaign must be supplied together");
1427
+ if (args.neutralizedOnHoldout && args.neutralizedOnHoldout.splitDigest !== args.baselineOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: neutralized campaign uses a different holdout split");
1428
+ if (holdoutDeferred && args.neutralizedOnHoldout) throw new Error("buildLoopProvenanceRecord: a deferred holdout cannot include a neutralized measurement");
1429
+ const holdoutMeasurement = holdoutDeferred ? { kind: "deferred" } : {
1430
+ kind: "measured",
1431
+ baseline: meanHoldoutComposite(args.baselineOnHoldout),
1432
+ winner: meanHoldoutComposite(args.winnerOnHoldout),
1433
+ ...args.neutralizedOnHoldout ? { neutralized: meanHoldoutComposite(args.neutralizedOnHoldout) } : {}
1434
+ };
1435
+ const diff = surfaceContentHash(args.baselineSurface) === surfaceContentHash(args.winnerSurface) ? "" : renderSurfaceDiff(args.winnerSurface, args.baselineSurface);
1436
+ const recordWithoutDigest = {
1437
+ schema: "tangle.loop-provenance",
1438
+ runId: args.runId,
1439
+ runDir: args.runDir,
1440
+ timestamp: args.timestamp,
1441
+ baselineContentHash: surfaceContentHash(args.baselineSurface),
1442
+ winnerContentHash: surfaceContentHash(args.winnerSurface),
1443
+ diff,
1444
+ candidates,
1445
+ evidence: {
1446
+ search: {
1447
+ splitDigest: args.baselineSearchCampaign.splitDigest,
1448
+ baselineCampaignDigest: campaignMeasurementDigest(args.baselineSearchCampaign)
1449
+ },
1450
+ holdout: {
1451
+ splitDigest: args.baselineOnHoldout.splitDigest,
1452
+ baselineCampaignDigest: campaignMeasurementDigest(args.baselineOnHoldout),
1453
+ winnerCampaignDigest: campaignMeasurementDigest(args.winnerOnHoldout),
1454
+ ...args.neutralizedSurface && args.neutralizedOnHoldout && holdoutMeasurement.kind === "measured" && holdoutMeasurement.neutralized !== void 0 ? { neutralized: {
1455
+ contentHash: surfaceContentHash(args.neutralizedSurface),
1456
+ campaignDigest: campaignMeasurementDigest(args.neutralizedOnHoldout),
1457
+ composite: holdoutMeasurement.neutralized,
1458
+ lift: holdoutMeasurement.neutralized - holdoutMeasurement.baseline
1459
+ } } : {}
1460
+ },
1461
+ costReceiptsDigest: canonicalDigest([...args.costReceipts].sort((left, right) => compareCodeUnits(left.callId, right.callId)))
1462
+ },
1463
+ baselineSearchComposite,
1464
+ gate: {
1465
+ decision: args.gate.decision,
1466
+ reasons: args.gate.reasons,
1467
+ ...args.gate.delta === void 0 ? {} : { delta: args.gate.delta },
1468
+ contributingGates: args.gate.contributingGates.map((g) => ({
1469
+ name: g.name,
1470
+ status: g.status,
1471
+ detail: durableGateDetail(g.detail)
1472
+ }))
1473
+ },
1474
+ ...holdoutMeasurement.kind === "deferred" ? { holdout: "deferred" } : {
1475
+ baselineHoldoutComposite: holdoutMeasurement.baseline,
1476
+ winnerHoldoutComposite: holdoutMeasurement.winner,
1477
+ heldOutLift: holdoutMeasurement.winner - holdoutMeasurement.baseline
1478
+ },
1479
+ backend: {
1480
+ verdict: integrity.verdict,
1481
+ workerCallCount: integrity.totalRecords,
1482
+ models,
1483
+ totalInputTokens: integrity.totalInputTokens,
1484
+ totalOutputTokens: integrity.totalOutputTokens,
1485
+ totalCostUsd: integrity.totalCostUsd
1486
+ },
1487
+ totalCostUsd: args.totalCostUsd,
1488
+ totalDurationMs: args.totalDurationMs
1489
+ };
1490
+ if (args.optimizationMethod) recordWithoutDigest.optimizationMethod = durableOptimizationMethod(args.optimizationMethod);
1491
+ if (args.winnerLabel) recordWithoutDigest.winnerLabel = args.winnerLabel;
1492
+ if (args.winnerRationale) recordWithoutDigest.winnerRationale = args.winnerRationale;
1493
+ return {
1494
+ ...recordWithoutDigest,
1495
+ recordDigest: canonicalDigest(recordWithoutDigest)
1496
+ };
1497
+ }
1498
+ function durableOptimizationMethod(value) {
1499
+ if (!value || typeof value !== "object" || typeof value.name !== "string" || !value.name.trim() || value.name.trim() !== value.name) throw new Error("buildLoopProvenanceRecord: optimization method name is invalid");
1500
+ if (!value.cost || !Number.isFinite(value.cost.totalCostUsd) || value.cost.totalCostUsd < 0 || typeof value.cost.accountingComplete !== "boolean" || !Array.isArray(value.cost.incompleteReasons) || value.cost.incompleteReasons.some((reason) => typeof reason !== "string" || !reason.trim()) || value.cost.accountingComplete !== (value.cost.incompleteReasons.length === 0)) throw new Error("buildLoopProvenanceRecord: optimization method cost is invalid");
1501
+ if (value.durationMs !== void 0 && (!Number.isFinite(value.durationMs) || value.durationMs < 0)) throw new Error("buildLoopProvenanceRecord: optimization method duration is invalid");
1502
+ try {
1503
+ return JSON.parse(canonicalString(value));
1504
+ } catch (cause) {
1505
+ throw new Error("buildLoopProvenanceRecord: optimization method data must be canonical JSON", { cause });
1506
+ }
1507
+ }
1508
+ /** Digest the exact campaign fields that can affect a measured comparison. */
1509
+ function campaignMeasurementDigest(campaign) {
1510
+ assertCampaignSplitIdentity(campaign.scenarios, campaign.reps, campaign.splitDigest);
1511
+ return canonicalDigest({
1512
+ schema: "tangle.campaign-measurement",
1513
+ manifestHash: campaign.manifestHash,
1514
+ splitDigest: campaign.splitDigest,
1515
+ seed: campaign.seed,
1516
+ reps: campaign.reps,
1517
+ runDir: campaign.runDir,
1518
+ scenarios: campaign.scenarios,
1519
+ cells: [...campaign.cells].sort((left, right) => compareCodeUnits(left.cellId, right.cellId)).map((cell) => ({
1520
+ manifestHash: cell.manifestHash ?? null,
1521
+ cellId: cell.cellId,
1522
+ scenarioId: cell.scenarioId,
1523
+ rep: cell.rep,
1524
+ generation: cell.generation ?? null,
1525
+ judgeScores: cell.judgeScores,
1526
+ costUsd: cell.costUsd,
1527
+ costProvenance: cell.costProvenance,
1528
+ costCallIds: [...cell.costCallIds ?? []].sort(),
1529
+ tokenUsage: cell.tokenUsage,
1530
+ resolvedModels: [...cell.resolvedModels ?? []].sort(),
1531
+ resolvedModel: cell.resolvedModel ?? null,
1532
+ durationMs: cell.durationMs,
1533
+ seed: cell.seed,
1534
+ cached: cell.cached,
1535
+ errorStage: cell.errorStage ?? null,
1536
+ errorJudge: cell.errorJudge ?? null,
1537
+ error: cell.error ?? null
1538
+ }))
1539
+ });
1540
+ }
1541
+ /** Recompute and validate the self-addressed durable record. */
1542
+ function verifyLoopProvenanceRecord(record) {
1543
+ if (record.schema !== "tangle.loop-provenance") throw new Error("loop provenance has an unsupported schema");
1544
+ const { recordDigest, ...recordWithoutDigest } = record;
1545
+ if (recordDigest !== canonicalDigest(recordWithoutDigest)) throw new Error("loop provenance record digest does not match its contents");
1546
+ assertGateContributions(record.gate?.contributingGates, "loop provenance");
1547
+ return record;
1548
+ }
1549
+ /** SHA-256 over the RFC 8785 canonical JSON of `value`. Throws
1550
+ * `LedgerCanonicalizationError` for a value with no canonical form. */
1551
+ function canonicalDigest(value) {
1552
+ return hashCanonical(value);
1553
+ }
1554
+ function durableGateDetail(detail) {
1555
+ if (detail === void 0) return null;
1556
+ try {
1557
+ return JSON.parse(canonicalString(detail));
1558
+ } catch (cause) {
1559
+ throw new Error("buildLoopProvenanceRecord: gate detail must be canonical JSON", { cause });
1560
+ }
1561
+ }
1562
+ function assertGateContributions(value, source) {
1563
+ if (!Array.isArray(value)) throw new Error(`${source}: gate contributingGates must be an array`);
1564
+ const statuses = /* @__PURE__ */ new Set([
1565
+ "pass",
1566
+ "fail",
1567
+ "not_evaluated"
1568
+ ]);
1569
+ for (const [index, contribution] of value.entries()) {
1570
+ if (!contribution || typeof contribution !== "object") throw new Error(`${source}: gate contribution ${index} must be an object`);
1571
+ const item = contribution;
1572
+ if (typeof item.name !== "string" || item.name.length === 0) throw new Error(`${source}: gate contribution ${index} must have a non-empty name`);
1573
+ if (!statuses.has(String(item.status))) throw new Error(`${source}: gate contribution '${item.name}' must have status pass, fail, or not_evaluated`);
1574
+ if ("passed" in item) throw new Error(`${source}: gate contribution '${item.name}' uses obsolete passed; use status instead`);
1575
+ }
1576
+ }
1577
+ function validateCandidateMeasurement(candidate, expectedParentHash, expectedParentComposite, promoted) {
1578
+ if (!candidate.parentSurfaceHash || !/^[a-f0-9]{16}$/.test(candidate.parentSurfaceHash)) throw new Error("buildLoopProvenanceRecord: parentSurfaceHash must be 16 lowercase hex characters");
1579
+ if (candidate.parentSurfaceHash !== expectedParentHash) throw new Error("buildLoopProvenanceRecord: candidate parent does not match the incumbent");
1580
+ if (candidate.parentComposite === void 0 || !Number.isFinite(candidate.parentComposite) || Math.abs(candidate.parentComposite - expectedParentComposite) > 1e-12) throw new Error("buildLoopProvenanceRecord: candidate parentComposite does not match the incumbent");
1581
+ if (candidate.observedDeltaFromParent !== void 0) {
1582
+ if (!Number.isFinite(candidate.observedDeltaFromParent)) throw new Error("buildLoopProvenanceRecord: observedDeltaFromParent must be finite");
1583
+ if (candidate.eligibleForPromotion !== true) throw new Error("buildLoopProvenanceRecord: observedDeltaFromParent requires a complete eligible candidate and parentSurfaceHash");
1584
+ }
1585
+ const coverage = candidate.coverage;
1586
+ if (!Number.isSafeInteger(coverage.expectedCells) || coverage.expectedCells <= 0 || !Number.isSafeInteger(coverage.scorableCells) || coverage.scorableCells < 0 || coverage.scorableCells > coverage.expectedCells) throw new Error("buildLoopProvenanceRecord: invalid candidate coverage denominator");
1587
+ const unscorableIds = /* @__PURE__ */ new Set();
1588
+ for (const failure of coverage.unscorableCells) {
1589
+ if (typeof failure.cellId !== "string" || failure.cellId.length === 0 || typeof failure.reason !== "string" || failure.reason.length === 0 || unscorableIds.has(failure.cellId)) throw new Error("buildLoopProvenanceRecord: invalid candidate coverage failures");
1590
+ unscorableIds.add(failure.cellId);
1591
+ }
1592
+ if (coverage.expectedCells - coverage.scorableCells !== coverage.unscorableCells.length) throw new Error("buildLoopProvenanceRecord: candidate coverage counts do not match its failures");
1593
+ const complete = coverage.scorableCells === coverage.expectedCells && coverage.unscorableCells.length === 0;
1594
+ if (candidate.eligibleForPromotion !== complete) throw new Error("buildLoopProvenanceRecord: candidate eligibility contradicts its coverage receipt");
1595
+ if (complete) {
1596
+ if (candidate.composite === null || !Number.isFinite(candidate.composite)) throw new Error("buildLoopProvenanceRecord: complete candidate composite must be finite");
1597
+ if (candidate.observedDeltaFromParent === void 0) throw new Error("buildLoopProvenanceRecord: complete candidate is missing observedDeltaFromParent");
1598
+ const recomputed = candidate.composite - candidate.parentComposite;
1599
+ if (Math.abs(candidate.observedDeltaFromParent - recomputed) > 1e-12) throw new Error("buildLoopProvenanceRecord: observed delta does not match measured scores");
1600
+ } else {
1601
+ if (candidate.composite !== null && !Number.isFinite(candidate.composite)) throw new Error("buildLoopProvenanceRecord: candidate composite must be finite or null");
1602
+ if (candidate.observedDeltaFromParent !== void 0) throw new Error("buildLoopProvenanceRecord: incomplete candidate cannot carry observed delta");
1603
+ }
1604
+ if (promoted && (!complete || (candidate.observedDeltaFromParent ?? 0) <= 0)) throw new Error("buildLoopProvenanceRecord: promoted candidate must improve the incumbent");
1605
+ }
1606
+ function hashId(parts) {
1607
+ return createHash("sha256").update(parts.join(":")).digest("hex");
1608
+ }
1609
+ /**
1610
+ * Build the loop's OTLP-ingestable spans from a provenance record. One root
1611
+ * span per loop (`tangle.runId`), one span per generation, one span per
1612
+ * candidate (carrying its surfaceHash + label), and one span for the gate
1613
+ * decision (carrying reasons + delta + lift). Candidate + gate spans pivot on
1614
+ * the same `tangle.runId` / `tangle.generation` attributes `/adapters/otel`
1615
+ * reads, so the hosted collector reconstructs the full tree.
1616
+ *
1617
+ * Times are synthesized monotonically off a single base so the span tree is
1618
+ * orderable; the substrate does not retain per-candidate wall-clock starts.
1619
+ */
1620
+ function loopProvenanceSpans(record, opts = {}) {
1621
+ const traceId = hashId(["trace", record.runId]).slice(0, 32);
1622
+ const baseTimeMs = opts.baseTimeMs ?? (Date.parse(record.timestamp) || Date.now());
1623
+ const durationMs = Math.max(1, record.totalDurationMs);
1624
+ if (!Number.isSafeInteger(baseTimeMs) || baseTimeMs < 0) throw new RangeError("loop provenance baseTimeMs must be a non-negative safe integer");
1625
+ if (!Number.isSafeInteger(durationMs)) throw new RangeError("loop provenance duration must be a safe integer number of milliseconds");
1626
+ const baseTime = BigInt(baseTimeMs);
1627
+ const baseNano = (baseTime * 1000000n).toString();
1628
+ const endNano = ((baseTime + BigInt(durationMs)) * 1000000n).toString();
1629
+ const spans = [];
1630
+ const rootSpanId = hashId(["root", record.runId]).slice(0, 16);
1631
+ const rootAttributes = {
1632
+ "tangle.runId": record.runId,
1633
+ "tangle.runDir": record.runDir,
1634
+ "tangle.baselineContentHash": record.baselineContentHash,
1635
+ "tangle.winnerContentHash": record.winnerContentHash,
1636
+ "tangle.baselineSearchComposite": record.baselineSearchComposite,
1637
+ "tangle.gateDecision": record.gate.decision,
1638
+ "tangle.backendVerdict": record.backend.verdict,
1639
+ "tangle.workerCallCount": record.backend.workerCallCount,
1640
+ "tangle.totalCostUsd": record.totalCostUsd
1641
+ };
1642
+ if (record.heldOutLift !== void 0) rootAttributes["tangle.heldOutLift"] = record.heldOutLift;
1643
+ if (record.holdout) rootAttributes["tangle.holdout"] = record.holdout;
1644
+ spans.push({
1645
+ traceId,
1646
+ spanId: rootSpanId,
1647
+ name: "improvement-loop",
1648
+ startTimeUnixNano: baseNano,
1649
+ endTimeUnixNano: endNano,
1650
+ attributes: rootAttributes,
1651
+ status: { code: "OK" },
1652
+ "tangle.runId": record.runId
1653
+ });
1654
+ const byGen = /* @__PURE__ */ new Map();
1655
+ for (const c of record.candidates) {
1656
+ const arr = byGen.get(c.generation) ?? [];
1657
+ arr.push(c);
1658
+ byGen.set(c.generation, arr);
1659
+ }
1660
+ for (const [generation, cands] of [...byGen.entries()].sort((a, b) => a[0] - b[0])) {
1661
+ const genSpanId = hashId([
1662
+ "gen",
1663
+ record.runId,
1664
+ String(generation)
1665
+ ]).slice(0, 16);
1666
+ const measuredComposites = cands.flatMap((candidate) => candidate.composite === null ? [] : [candidate.composite]);
1667
+ spans.push({
1668
+ traceId,
1669
+ spanId: genSpanId,
1670
+ parentSpanId: rootSpanId,
1671
+ name: `generation-${generation}`,
1672
+ startTimeUnixNano: baseNano,
1673
+ endTimeUnixNano: endNano,
1674
+ attributes: {
1675
+ "tangle.runId": record.runId,
1676
+ "tangle.generation": generation,
1677
+ "tangle.populationSize": cands.length,
1678
+ ...measuredComposites.length > 0 ? { "tangle.bestComposite": Math.max(...measuredComposites) } : {}
1679
+ },
1680
+ "tangle.runId": record.runId,
1681
+ "tangle.generation": generation
1682
+ });
1683
+ for (let i = 0; i < cands.length; i++) {
1684
+ const c = cands[i];
1685
+ const candSpanId = hashId([
1686
+ "cand",
1687
+ record.runId,
1688
+ String(generation),
1689
+ c.surfaceHash
1690
+ ]).slice(0, 16);
1691
+ const attributes = {
1692
+ "tangle.runId": record.runId,
1693
+ "tangle.generation": generation,
1694
+ "tangle.surfaceHash": c.surfaceHash,
1695
+ "tangle.contentHash": c.contentHash,
1696
+ "tangle.parentSurfaceHash": c.parentSurfaceHash,
1697
+ "tangle.parentComposite": c.parentComposite,
1698
+ "tangle.eligibleForPromotion": c.eligibleForPromotion,
1699
+ "tangle.expectedCells": c.coverage.expectedCells,
1700
+ "tangle.scorableCells": c.coverage.scorableCells,
1701
+ "tangle.unscorableCells": c.coverage.unscorableCells.length,
1702
+ "tangle.promoted": c.promoted
1703
+ };
1704
+ if (c.composite !== null) attributes["tangle.composite"] = c.composite;
1705
+ if (c.observedDeltaFromParent !== void 0) attributes["tangle.observedDeltaFromParent"] = c.observedDeltaFromParent;
1706
+ if (c.label) attributes["tangle.candidateLabel"] = c.label;
1707
+ if (c.rationale) attributes["tangle.candidateRationale"] = c.rationale;
1708
+ spans.push({
1709
+ traceId,
1710
+ spanId: candSpanId,
1711
+ parentSpanId: genSpanId,
1712
+ name: `candidate-${c.surfaceHash}`,
1713
+ startTimeUnixNano: baseNano,
1714
+ endTimeUnixNano: endNano,
1715
+ attributes,
1716
+ "tangle.runId": record.runId,
1717
+ "tangle.generation": generation
1718
+ });
1719
+ }
1720
+ }
1721
+ const gateSpanId = hashId(["gate", record.runId]).slice(0, 16);
1722
+ const gateAttributes = {
1723
+ "tangle.runId": record.runId,
1724
+ "tangle.gateDecision": record.gate.decision,
1725
+ "tangle.gateReasons": JSON.stringify(record.gate.reasons)
1726
+ };
1727
+ const gateDelta = record.gate.delta ?? record.heldOutLift;
1728
+ if (gateDelta !== void 0) gateAttributes["tangle.gateDelta"] = gateDelta;
1729
+ if (record.heldOutLift !== void 0) gateAttributes["tangle.heldOutLift"] = record.heldOutLift;
1730
+ if (record.baselineHoldoutComposite !== void 0) gateAttributes["tangle.baselineHoldoutComposite"] = record.baselineHoldoutComposite;
1731
+ if (record.winnerHoldoutComposite !== void 0) gateAttributes["tangle.winnerHoldoutComposite"] = record.winnerHoldoutComposite;
1732
+ if (record.holdout) gateAttributes["tangle.holdout"] = record.holdout;
1733
+ spans.push({
1734
+ traceId,
1735
+ spanId: gateSpanId,
1736
+ parentSpanId: rootSpanId,
1737
+ name: "gate-decision",
1738
+ startTimeUnixNano: endNano,
1739
+ endTimeUnixNano: endNano,
1740
+ attributes: gateAttributes,
1741
+ status: { code: "OK" },
1742
+ "tangle.runId": record.runId
1743
+ });
1744
+ return spans;
1745
+ }
1746
+ /** Canonical durable paths under the run dir. */
1747
+ function provenanceRecordPath(runDir) {
1748
+ return join(runDir, "loop-provenance.json");
1749
+ }
1750
+ /**
1751
+ * Canonical path for the durable OTLP spans JSONL file under a loop run directory.
1752
+ */
1753
+ function provenanceSpansPath(runDir) {
1754
+ return join(runDir, "loop-provenance-spans.jsonl");
1755
+ }
1756
+ /** Snapshot a held-out campaign into the hosted `EvalRunGenerationSnapshot`
1757
+ * shape — per-cell composite + per-judge dimensions, aggregate mean, cost,
1758
+ * duration. The dashboard renders these as the baseline → winner comparison. */
1759
+ function snapshotFromHoldout(index, surfaceHash, surface, campaign) {
1760
+ return {
1761
+ index,
1762
+ surfaceHash,
1763
+ surface,
1764
+ cells: campaign.cells.map((cell) => {
1765
+ const execution = campaignCellExecutionEvidence(cell);
1766
+ const quality = projectCampaignCellQuality(cell);
1767
+ const score = {
1768
+ scenarioId: cell.scenarioId,
1769
+ rep: cell.rep,
1770
+ compositeMean: quality.score ?? null,
1771
+ dimensions: quality.judgeScores?.perJudge ?? {},
1772
+ terminalOutcome: execution.terminalOutcome,
1773
+ executionErrorCount: execution.executionErrorCount ?? null
1774
+ };
1775
+ if (cell.error) score.errorMessage = cell.error;
1776
+ return score;
1777
+ }),
1778
+ compositeMean: campaignMeanCompositeOrNull(campaign),
1779
+ costUsd: campaign.aggregates.cost.totalCostUsd,
1780
+ durationMs: campaign.durationMs
1781
+ };
1782
+ }
1783
+ /** Build the hosted `EvalRunEvent` from the loop args + record — baseline +
1784
+ * winner snapshots, gate decision, held-out lift, cost, duration. Shipped to
1785
+ * `/v1/ingest/eval-runs` so the run appears in the dashboard's run list (the
1786
+ * trace spans, shipped separately, back the per-candidate drill-down). */
1787
+ function buildEvalRunEvent(args, record) {
1788
+ return {
1789
+ runId: args.runId,
1790
+ runDir: args.runDir,
1791
+ timestamp: args.timestamp,
1792
+ status: "finished",
1793
+ labels: {},
1794
+ baseline: snapshotFromHoldout(0, record.baselineContentHash, args.baselineSurface, args.baselineOnHoldout),
1795
+ generations: [snapshotFromHoldout(1, record.winnerContentHash, args.winnerSurface, args.winnerOnHoldout)],
1796
+ gateDecision: args.gate.decision,
1797
+ ...record.heldOutLift !== void 0 ? { holdoutLift: record.heldOutLift } : {},
1798
+ totalCostUsd: args.totalCostUsd,
1799
+ totalDurationMs: args.totalDurationMs
1800
+ };
1801
+ }
1802
+ /**
1803
+ * Build the provenance record + OTel spans and persist them durably under the
1804
+ * run dir (and ship spans to a hosted collector when one is wired). Returns
1805
+ * both artifacts so the caller can assert on / re-derive from them.
1806
+ *
1807
+ * Fail-loud: the durable write throws on storage failure (a swallowed write is
1808
+ * exactly the "emitted but lost" failure this closes). The hosted span ship is
1809
+ * the one best-effort leg — its failure is logged, not thrown, so an offline
1810
+ * collector never fails the loop (the durable artifact is the source of truth).
1811
+ */
1812
+ async function emitLoopProvenance(args) {
1813
+ const record = buildLoopProvenanceRecord(args);
1814
+ const spans = loopProvenanceSpans(record);
1815
+ args.storage.ensureDir(args.runDir);
1816
+ const recordPath = provenanceRecordPath(args.runDir);
1817
+ const spansPath = provenanceSpansPath(args.runDir);
1818
+ args.storage.write(recordPath, JSON.stringify(record, null, 2));
1819
+ args.storage.write(spansPath, spans.map((s) => JSON.stringify(s)).join("\n"));
1820
+ if (args.hostedClient) {
1821
+ try {
1822
+ await args.hostedClient.ingestEvalRun(buildEvalRunEvent(args, record));
1823
+ } catch (err) {
1824
+ const msg = err instanceof Error ? err.message : String(err);
1825
+ console.warn(`[agent-eval] hosted eval-run ingest failed (continuing): ${msg}`);
1826
+ }
1827
+ try {
1828
+ await args.hostedClient.ingestTraces(spans);
1829
+ } catch (err) {
1830
+ const msg = err instanceof Error ? err.message : String(err);
1831
+ console.warn(`[agent-eval] provenance span ingest failed (continuing): ${msg}`);
1832
+ }
1833
+ }
1834
+ return {
1835
+ record,
1836
+ spans,
1837
+ recordPath,
1838
+ spansPath
1839
+ };
1840
+ }
1841
+ //#endregion
1842
+ //#region src/attestation.ts
1843
+ /**
1844
+ * Reproducibility attestation for any serializable report object.
1845
+ *
1846
+ * `attest()` binds a report to its content address (sha-256 over canonical
1847
+ * JSON) AND binds that address to the provenance needed to reproduce it:
1848
+ * model versions, seeds, price-table hash, code SHA, inputs hash. The outer
1849
+ * `envelopeHash` prevents provenance from being rewritten while leaving the
1850
+ * report hash valid.
1851
+ *
1852
+ * Layering: content-addressing is the substrate's job; cryptographic SIGNING
1853
+ * (who vouches for the attestation, key management, transparency logs) is the
1854
+ * consumer's layer on top. An `AttestedReport` is a stable byte-identical
1855
+ * payload a consumer can sign — the substrate never holds keys.
1856
+ *
1857
+ * Generic by design: the report parameter is ANY value `canonicalJson`
1858
+ * accepts (campaign results, fuzz capsules, scorecards, cost ledgers). Do not
1859
+ * couple this module to a specific report schema.
1860
+ */
1861
+ /** Hash scheme identifier carried by every attestation. A verifier rejects
1862
+ * unknown algorithms instead of guessing. */
1863
+ const ATTESTATION_ALGORITHM = "sha256/canonical-json";
1864
+ function envelopeMaterial(reportHash, provenance, algorithm) {
1865
+ return {
1866
+ reportHash,
1867
+ provenance,
1868
+ algorithm
1869
+ };
1870
+ }
1871
+ /**
1872
+ * Content-address a report and bind it to its provenance. Throws (via
1873
+ * `canonicalJson`) if the report or provenance contains undefined / function /
1874
+ * symbol / non-finite numbers — an attestation that cannot be unambiguously
1875
+ * serialized cannot be trusted.
1876
+ */
1877
+ function attest(report, provenance) {
1878
+ const reportHash = contentHash(report);
1879
+ const algorithm = ATTESTATION_ALGORITHM;
1880
+ return {
1881
+ reportHash,
1882
+ provenance,
1883
+ algorithm,
1884
+ envelopeHash: contentHash(envelopeMaterial(reportHash, provenance, algorithm))
1885
+ };
1886
+ }
1887
+ /**
1888
+ * Verify a report against its attestation. Returns a typed outcome rather
1889
+ * than throwing: an unverifiable report (e.g. one that no longer
1890
+ * canonicalizes) is a verification failure with the cause in `reason`, not a
1891
+ * crash — verifiers run in pipelines that must record WHY, not die.
1892
+ *
1893
+ * Legacy attestations without `envelopeHash` remain readable, but verification
1894
+ * explicitly marks their provenance as unbound so a promotion path can refuse
1895
+ * them instead of accidentally treating old metadata as cryptographic proof.
1896
+ */
1897
+ function verifyAttestation(report, attested) {
1898
+ if (attested.algorithm !== "sha256/canonical-json") return {
1899
+ valid: false,
1900
+ reason: `unknown algorithm '${attested.algorithm}' — this verifier only checks '${ATTESTATION_ALGORITHM}'`
1901
+ };
1902
+ let recomputed;
1903
+ try {
1904
+ recomputed = contentHash(report);
1905
+ } catch (err) {
1906
+ return {
1907
+ valid: false,
1908
+ reason: `report is not canonicalizable: ${err instanceof Error ? err.message : String(err)}`
1909
+ };
1910
+ }
1911
+ if (recomputed !== attested.reportHash) return {
1912
+ valid: false,
1913
+ reason: `report hash mismatch: attested ${attested.reportHash}, recomputed ${recomputed}`
1914
+ };
1915
+ if (attested.envelopeHash === void 0) return {
1916
+ valid: true,
1917
+ legacyUnboundProvenance: true
1918
+ };
1919
+ let envelopeHash;
1920
+ try {
1921
+ envelopeHash = contentHash(envelopeMaterial(attested.reportHash, attested.provenance, attested.algorithm));
1922
+ } catch (err) {
1923
+ return {
1924
+ valid: false,
1925
+ reason: `attestation provenance is not canonicalizable: ${err instanceof Error ? err.message : String(err)}`
1926
+ };
1927
+ }
1928
+ if (envelopeHash !== attested.envelopeHash) return {
1929
+ valid: false,
1930
+ reason: `attestation envelope hash mismatch: attested ${attested.envelopeHash}, recomputed ${envelopeHash}`
1931
+ };
1932
+ return { valid: true };
1933
+ }
1934
+ //#endregion
1935
+ //#region src/experiment/evidence-receipt.ts
1936
+ /**
1937
+ * Evidence receipts are the join between Runtime execution and Eval promotion.
1938
+ *
1939
+ * Runtime says what actually ran. Eval says what independently measured it.
1940
+ * This receipt binds those worlds without making either package import the other:
1941
+ * stable pursuit/run identity, exact candidate/evaluator/environment/input/output
1942
+ * content identities, and the authority class that produced the observation.
1943
+ *
1944
+ * The payload is attested with agent-eval's existing canonical report attestation.
1945
+ * Signing/key management deliberately remains outside this substrate; consumers may
1946
+ * sign the byte-stable receipt or anchor its attestation in a transparency log.
1947
+ */
1948
+ const EVIDENCE_RECEIPT_VERSION = "1.0.0";
1949
+ /** Closed promotion vocabulary. A typo or unknown future kind is never independent by default. */
1950
+ const EVIDENCE_AUTHORITY_KINDS = [
1951
+ "candidate-self-report",
1952
+ "independent-evaluator",
1953
+ "independent-replication",
1954
+ "human-review",
1955
+ "production-canary"
1956
+ ];
1957
+ const INDEPENDENT_EVIDENCE_AUTHORITY_KINDS = [
1958
+ "independent-evaluator",
1959
+ "independent-replication",
1960
+ "human-review",
1961
+ "production-canary"
1962
+ ];
1963
+ /**
1964
+ * Mint a content-attested evidence receipt. Required identity fields are deliberately
1965
+ * non-optional: unknown evidence stays unknown and cannot accidentally look certified.
1966
+ * The attestation's input provenance must equal the receipt commitment so two competing
1967
+ * descriptions of the evaluated population cannot coexist inside one valid receipt.
1968
+ */
1969
+ function createEvidenceReceipt(input, provenance) {
1970
+ const inputSetCommitment = requiredIdentity(input.inputSetCommitment, "inputSetCommitment");
1971
+ assertInputCommitment(provenance, inputSetCommitment);
1972
+ const binding = Object.freeze({
1973
+ schemaVersion: EVIDENCE_RECEIPT_VERSION,
1974
+ pursuitId: requiredIdentity(input.pursuitId, "pursuitId"),
1975
+ runId: requiredIdentity(input.runId, "runId"),
1976
+ candidateDigest: requiredIdentity(input.candidateDigest, "candidateDigest"),
1977
+ evaluatorDigest: requiredIdentity(input.evaluatorDigest, "evaluatorDigest"),
1978
+ environmentDigest: requiredIdentity(input.environmentDigest, "environmentDigest"),
1979
+ inputSetCommitment,
1980
+ outputDigest: requiredIdentity(input.outputDigest, "outputDigest"),
1981
+ resultDigest: requiredIdentity(input.resultDigest, "resultDigest"),
1982
+ authority: requiredAuthority(input.authority),
1983
+ ...input.experimentDigest === void 0 ? {} : { experimentDigest: requiredIdentity(input.experimentDigest, "experimentDigest") },
1984
+ ...input.observerDigest === void 0 ? {} : { observerDigest: requiredIdentity(input.observerDigest, "observerDigest") }
1985
+ });
1986
+ return Object.freeze({
1987
+ binding,
1988
+ attestation: attest(binding, provenance)
1989
+ });
1990
+ }
1991
+ /**
1992
+ * Verify promotion-grade evidence. Generic report attestation keeps a legacy read path,
1993
+ * but an EvidenceReceipt never accepts unbound provenance: changing the evaluator code,
1994
+ * model versions, input commitment provenance, or creation record must invalidate the
1995
+ * evidence rather than merely annotating it as legacy.
1996
+ */
1997
+ function verifyEvidenceReceipt(receipt) {
1998
+ if (receipt.binding.schemaVersion !== "1.0.0") return {
1999
+ valid: false,
2000
+ reason: `unsupported evidence receipt version '${receipt.binding.schemaVersion}'`
2001
+ };
2002
+ try {
2003
+ for (const [field, value] of Object.entries({
2004
+ pursuitId: receipt.binding.pursuitId,
2005
+ runId: receipt.binding.runId,
2006
+ candidateDigest: receipt.binding.candidateDigest,
2007
+ evaluatorDigest: receipt.binding.evaluatorDigest,
2008
+ environmentDigest: receipt.binding.environmentDigest,
2009
+ inputSetCommitment: receipt.binding.inputSetCommitment,
2010
+ outputDigest: receipt.binding.outputDigest,
2011
+ resultDigest: receipt.binding.resultDigest
2012
+ })) requiredIdentity(value, field);
2013
+ requiredAuthority(receipt.binding.authority);
2014
+ assertInputCommitment(receipt.attestation.provenance, receipt.binding.inputSetCommitment);
2015
+ } catch (error) {
2016
+ return {
2017
+ valid: false,
2018
+ reason: error instanceof Error ? error.message : String(error)
2019
+ };
2020
+ }
2021
+ const verification = verifyAttestation(receipt.binding, receipt.attestation);
2022
+ if (!verification.valid) return verification;
2023
+ if (verification.legacyUnboundProvenance === true || receipt.attestation.envelopeHash === void 0) return {
2024
+ valid: false,
2025
+ reason: "evidence receipt provenance is not bound by an attestation envelope"
2026
+ };
2027
+ return { valid: true };
2028
+ }
2029
+ /**
2030
+ * Promotion may choose a stricter policy, but this primitive makes the basic separation
2031
+ * explicit: only a recognized independent authority is independent. Unknown/forged kinds
2032
+ * and candidate self-reports both return false.
2033
+ */
2034
+ function isIndependentEvidence(receipt) {
2035
+ return INDEPENDENT_EVIDENCE_AUTHORITY_KINDS.includes(receipt.binding.authority?.kind);
2036
+ }
2037
+ function requiredAuthority(value) {
2038
+ if (typeof value !== "object" || value === null || Array.isArray(value)) throw new TypeError("evidence receipt: authority must be an object");
2039
+ const authority = value;
2040
+ if (typeof authority.kind !== "string" || !EVIDENCE_AUTHORITY_KINDS.includes(authority.kind)) throw new TypeError(`evidence receipt: unknown authority kind '${String(authority.kind)}'`);
2041
+ if (typeof authority.id !== "string") throw new TypeError("evidence receipt: authority.id must be a string");
2042
+ return Object.freeze({
2043
+ kind: authority.kind,
2044
+ id: requiredIdentity(authority.id, "authority.id")
2045
+ });
2046
+ }
2047
+ function assertInputCommitment(provenance, inputSetCommitment) {
2048
+ if (typeof provenance?.inputsHash !== "string" || provenance.inputsHash.trim().length === 0) throw new TypeError("evidence receipt: provenance.inputsHash is required");
2049
+ if (provenance.inputsHash.trim() !== inputSetCommitment) throw new TypeError("evidence receipt: provenance.inputsHash must equal binding.inputSetCommitment");
2050
+ }
2051
+ function requiredIdentity(value, field) {
2052
+ const normalized = value.trim();
2053
+ if (normalized.length === 0) throw new TypeError(`evidence receipt: ${field} must be non-empty`);
2054
+ return normalized;
2055
+ }
2056
+ //#endregion
2057
+ //#region src/experiment/campaign-evidence.ts
2058
+ /** Bind a complete measured campaign to its executed surface and actual outputs. */
2059
+ function createCampaignEvidenceReceipt(input) {
2060
+ const { campaign, surface, context } = input;
2061
+ assertCampaignSplitIdentity(campaign.scenarios, campaign.reps, campaign.splitDigest);
2062
+ const coverage = campaignCoverage(campaign.cells, campaign.scenarios, campaign.reps, true);
2063
+ if (campaign.scenarios.length === 0 || !coverage.complete) throw new Error("campaign evidence requires a complete, nonempty measurement");
2064
+ const { provenance, ...binding } = context;
2065
+ return createEvidenceReceipt({
2066
+ ...binding,
2067
+ runId: campaign.runDir,
2068
+ candidateDigest: surfaceContentHash(surface),
2069
+ inputSetCommitment: campaign.splitDigest,
2070
+ outputDigest: hashCanonical([...campaign.cells].sort((a, b) => a.cellId < b.cellId ? -1 : a.cellId > b.cellId ? 1 : 0).map((cell) => ({
2071
+ cellId: cell.cellId,
2072
+ artifact: cell.artifact
2073
+ }))),
2074
+ resultDigest: campaignMeasurementDigest(campaign)
2075
+ }, {
2076
+ ...provenance,
2077
+ inputsHash: campaign.splitDigest
2078
+ });
2079
+ }
2080
+ //#endregion
2081
+ export { campaignScenarioIdentity as $, detectScale as A, componentSurfaceIdentityMaterial as B, campaignMeanCompositeOrNull as C, recoverTruncatedJson as D, parseReflectionResponse as E, pairedDecisionShape as F, surfaceHashMatches as G, surfaceContentHash as H, minimumPairsForPairedDeltaTest as I, summarizeBackendIntegrity as J, BackendIntegrityError as K, pairedDeltaTest as L, heldoutSignificance as M, pairHoldout as N, powerPreflight as O, decidePairedPromotion as P, campaignCoverage as Q, assertCodeSurfaceIdentity as R, campaignMeanComposite as S, buildReflectionPrompt as T, surfaceDispatchRef as U, renderSurfaceDiff as V, surfaceHash as W, assertCampaignSplitIdentity as X, assertCampaignDesign as Y, assertCompleteCampaign as Z, provenanceRecordPath as _, createEvidenceReceipt as a, assertFiniteRankKey as b, ATTESTATION_ALGORITHM as c, buildLoopProvenanceRecord as d, campaignSplitDigest as et, campaignMeasurementDigest as f, loopProvenanceSpans as g, loopProvenanceArgsFromResult as h, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS as i, dimensionRegressions as j, TIE_WARN_FRACTION as k, attest as l, emitLoopProvenance as m, EVIDENCE_AUTHORITY_KINDS as n, formatCoverageFailures as nt, isIndependentEvidence as o, canonicalDigest as p, assertRealBackend as q, EVIDENCE_RECEIPT_VERSION as r, verifyEvidenceReceipt as s, createCampaignEvidenceReceipt as t, campaignSplitDigestFromIdentities as tt, verifyAttestation as u, provenanceSpansPath as v, compareRankKeys as w, campaignBreakdown as x, verifyLoopProvenanceRecord as y, codeSurfaceIdentityMaterial as z };
2082
+
2083
+ //# sourceMappingURL=campaign-evidence-D8DBLqLI.js.map