@tangle-network/agent-eval 0.120.0 → 0.120.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/package.json +1 -1
  3. package/dist/analyst/index.d.ts +0 -3111
  4. package/dist/analyst/index.js +0 -403
  5. package/dist/analyst/index.js.map +0 -1
  6. package/dist/authenticity/index.d.ts +0 -161
  7. package/dist/authenticity/index.js +0 -215
  8. package/dist/authenticity/index.js.map +0 -1
  9. package/dist/belief-state/index.d.ts +0 -1301
  10. package/dist/belief-state/index.js +0 -2152
  11. package/dist/belief-state/index.js.map +0 -1
  12. package/dist/benchmarks/index.d.ts +0 -974
  13. package/dist/benchmarks/index.js +0 -60
  14. package/dist/benchmarks/index.js.map +0 -1
  15. package/dist/builder-eval/index.d.ts +0 -695
  16. package/dist/builder-eval/index.js +0 -366
  17. package/dist/builder-eval/index.js.map +0 -1
  18. package/dist/campaign/index.d.ts +0 -7454
  19. package/dist/campaign/index.js +0 -272
  20. package/dist/campaign/index.js.map +0 -1
  21. package/dist/chunk-32BZXMSO.js +0 -3878
  22. package/dist/chunk-32BZXMSO.js.map +0 -1
  23. package/dist/chunk-3A246TSA.js +0 -998
  24. package/dist/chunk-3A246TSA.js.map +0 -1
  25. package/dist/chunk-3RF76KTD.js +0 -84
  26. package/dist/chunk-3RF76KTD.js.map +0 -1
  27. package/dist/chunk-3YYRZDON.js +0 -45
  28. package/dist/chunk-3YYRZDON.js.map +0 -1
  29. package/dist/chunk-4I2E3LLO.js +0 -1030
  30. package/dist/chunk-4I2E3LLO.js.map +0 -1
  31. package/dist/chunk-ARU2PZFM.js +0 -312
  32. package/dist/chunk-ARU2PZFM.js.map +0 -1
  33. package/dist/chunk-BOD4O7OF.js +0 -40
  34. package/dist/chunk-BOD4O7OF.js.map +0 -1
  35. package/dist/chunk-DPZAEKA6.js +0 -880
  36. package/dist/chunk-DPZAEKA6.js.map +0 -1
  37. package/dist/chunk-DTJ6QUQB.js +0 -131
  38. package/dist/chunk-DTJ6QUQB.js.map +0 -1
  39. package/dist/chunk-GGE4NNQT.js +0 -65
  40. package/dist/chunk-GGE4NNQT.js.map +0 -1
  41. package/dist/chunk-H5UD2323.js +0 -286
  42. package/dist/chunk-H5UD2323.js.map +0 -1
  43. package/dist/chunk-HHWE3POT.js +0 -94
  44. package/dist/chunk-HHWE3POT.js.map +0 -1
  45. package/dist/chunk-HKUCJ437.js +0 -787
  46. package/dist/chunk-HKUCJ437.js.map +0 -1
  47. package/dist/chunk-JHCHEVET.js +0 -274
  48. package/dist/chunk-JHCHEVET.js.map +0 -1
  49. package/dist/chunk-JHOJHHU7.js +0 -867
  50. package/dist/chunk-JHOJHHU7.js.map +0 -1
  51. package/dist/chunk-JM2SKQMS.js +0 -750
  52. package/dist/chunk-JM2SKQMS.js.map +0 -1
  53. package/dist/chunk-JN2FCO5W.js +0 -7958
  54. package/dist/chunk-JN2FCO5W.js.map +0 -1
  55. package/dist/chunk-K4DBDHLK.js +0 -158
  56. package/dist/chunk-K4DBDHLK.js.map +0 -1
  57. package/dist/chunk-K6N6XJJX.js +0 -306
  58. package/dist/chunk-K6N6XJJX.js.map +0 -1
  59. package/dist/chunk-MA6HLL3S.js +0 -65
  60. package/dist/chunk-MA6HLL3S.js.map +0 -1
  61. package/dist/chunk-MAZ26DC7.js +0 -99
  62. package/dist/chunk-MAZ26DC7.js.map +0 -1
  63. package/dist/chunk-MOXWMGPC.js +0 -577
  64. package/dist/chunk-MOXWMGPC.js.map +0 -1
  65. package/dist/chunk-NJC7U437.js +0 -626
  66. package/dist/chunk-NJC7U437.js.map +0 -1
  67. package/dist/chunk-NPCTHQIO.js +0 -91
  68. package/dist/chunk-NPCTHQIO.js.map +0 -1
  69. package/dist/chunk-ONWEPEDO.js +0 -57
  70. package/dist/chunk-ONWEPEDO.js.map +0 -1
  71. package/dist/chunk-OYZAPX5G.js +0 -1526
  72. package/dist/chunk-OYZAPX5G.js.map +0 -1
  73. package/dist/chunk-PC4UYEBM.js +0 -166
  74. package/dist/chunk-PC4UYEBM.js.map +0 -1
  75. package/dist/chunk-PICTDURQ.js +0 -766
  76. package/dist/chunk-PICTDURQ.js.map +0 -1
  77. package/dist/chunk-PJQFMIOX.js +0 -1182
  78. package/dist/chunk-PJQFMIOX.js.map +0 -1
  79. package/dist/chunk-PXD6ZFNY.js +0 -1107
  80. package/dist/chunk-PXD6ZFNY.js.map +0 -1
  81. package/dist/chunk-PXE2VKMX.js +0 -140
  82. package/dist/chunk-PXE2VKMX.js.map +0 -1
  83. package/dist/chunk-PZ5AY32C.js +0 -10
  84. package/dist/chunk-PZ5AY32C.js.map +0 -1
  85. package/dist/chunk-QBRSJK47.js +0 -622
  86. package/dist/chunk-QBRSJK47.js.map +0 -1
  87. package/dist/chunk-QWMPPZ3X.js +0 -550
  88. package/dist/chunk-QWMPPZ3X.js.map +0 -1
  89. package/dist/chunk-S3UZOQ5Y.js +0 -328
  90. package/dist/chunk-S3UZOQ5Y.js.map +0 -1
  91. package/dist/chunk-S5TT5R3L.js +0 -2668
  92. package/dist/chunk-S5TT5R3L.js.map +0 -1
  93. package/dist/chunk-T4SQEITX.js +0 -95
  94. package/dist/chunk-T4SQEITX.js.map +0 -1
  95. package/dist/chunk-TT4KNT67.js +0 -124
  96. package/dist/chunk-TT4KNT67.js.map +0 -1
  97. package/dist/chunk-U5CHZ5M3.js +0 -357
  98. package/dist/chunk-U5CHZ5M3.js.map +0 -1
  99. package/dist/chunk-ULOKLHIQ.js +0 -1937
  100. package/dist/chunk-ULOKLHIQ.js.map +0 -1
  101. package/dist/chunk-VI2UW6B6.js +0 -162
  102. package/dist/chunk-VI2UW6B6.js.map +0 -1
  103. package/dist/chunk-VQMK5FMP.js +0 -247
  104. package/dist/chunk-VQMK5FMP.js.map +0 -1
  105. package/dist/chunk-VSMTAMNK.js +0 -53
  106. package/dist/chunk-VSMTAMNK.js.map +0 -1
  107. package/dist/chunk-VZSRQ272.js +0 -149
  108. package/dist/chunk-VZSRQ272.js.map +0 -1
  109. package/dist/chunk-WW2A73HW.js +0 -159
  110. package/dist/chunk-WW2A73HW.js.map +0 -1
  111. package/dist/chunk-X4UCIOTZ.js +0 -136
  112. package/dist/chunk-X4UCIOTZ.js.map +0 -1
  113. package/dist/chunk-XDIRG3TO.js +0 -1266
  114. package/dist/chunk-XDIRG3TO.js.map +0 -1
  115. package/dist/chunk-XJYR7XFV.js +0 -317
  116. package/dist/chunk-XJYR7XFV.js.map +0 -1
  117. package/dist/chunk-ZET2UAYW.js +0 -89
  118. package/dist/chunk-ZET2UAYW.js.map +0 -1
  119. package/dist/chunk-ZZUXHH3R.js +0 -99
  120. package/dist/chunk-ZZUXHH3R.js.map +0 -1
  121. package/dist/cli.d.ts +0 -1
  122. package/dist/cli.js +0 -112
  123. package/dist/cli.js.map +0 -1
  124. package/dist/contract/index.d.ts +0 -4969
  125. package/dist/contract/index.js +0 -1653
  126. package/dist/contract/index.js.map +0 -1
  127. package/dist/control.d.ts +0 -1013
  128. package/dist/control.js +0 -34
  129. package/dist/control.js.map +0 -1
  130. package/dist/fuzz.d.ts +0 -759
  131. package/dist/fuzz.js +0 -714
  132. package/dist/fuzz.js.map +0 -1
  133. package/dist/hosted/index.d.ts +0 -730
  134. package/dist/hosted/index.js +0 -14
  135. package/dist/hosted/index.js.map +0 -1
  136. package/dist/index.d.ts +0 -16780
  137. package/dist/index.js +0 -12168
  138. package/dist/index.js.map +0 -1
  139. package/dist/matrix/index.d.ts +0 -155
  140. package/dist/matrix/index.js +0 -8
  141. package/dist/matrix/index.js.map +0 -1
  142. package/dist/meta-eval/index.d.ts +0 -1030
  143. package/dist/meta-eval/index.js +0 -417
  144. package/dist/meta-eval/index.js.map +0 -1
  145. package/dist/multishot/index.d.ts +0 -579
  146. package/dist/multishot/index.js +0 -589
  147. package/dist/multishot/index.js.map +0 -1
  148. package/dist/openapi.json +0 -992
  149. package/dist/pipelines/index.d.ts +0 -567
  150. package/dist/pipelines/index.js +0 -515
  151. package/dist/pipelines/index.js.map +0 -1
  152. package/dist/reporting.d.ts +0 -1277
  153. package/dist/reporting.js +0 -48
  154. package/dist/reporting.js.map +0 -1
  155. package/dist/rl.d.ts +0 -4092
  156. package/dist/rl.js +0 -1724
  157. package/dist/rl.js.map +0 -1
  158. package/dist/run-campaign-HNFPJET4.js +0 -14
  159. package/dist/run-campaign-HNFPJET4.js.map +0 -1
  160. package/dist/storyboard/index.d.ts +0 -279
  161. package/dist/storyboard/index.js +0 -767
  162. package/dist/storyboard/index.js.map +0 -1
  163. package/dist/trace-attributes.d.ts +0 -52
  164. package/dist/trace-attributes.js +0 -62
  165. package/dist/trace-attributes.js.map +0 -1
  166. package/dist/traces.d.ts +0 -2343
  167. package/dist/traces.js +0 -249
  168. package/dist/traces.js.map +0 -1
  169. package/dist/wire/index.d.ts +0 -1252
  170. package/dist/wire/index.js +0 -81
  171. package/dist/wire/index.js.map +0 -1
@@ -1,998 +0,0 @@
1
- import {
2
- summarizeBackendIntegrity
3
- } from "./chunk-XDIRG3TO.js";
4
- import {
5
- paretoChart
6
- } from "./chunk-DPZAEKA6.js";
7
- import {
8
- cohensD,
9
- pairedBootstrap,
10
- pairedMde,
11
- pairedTTest,
12
- pearsonR,
13
- requiredSampleSize,
14
- spearmanR
15
- } from "./chunk-PJQFMIOX.js";
16
- import {
17
- llmSpans
18
- } from "./chunk-ZET2UAYW.js";
19
- import {
20
- resolveRunCostProvenance
21
- } from "./chunk-S3UZOQ5Y.js";
22
-
23
- // src/contamination-guard.ts
24
- function checkCanaries(output, scenarios) {
25
- const leaks = [];
26
- for (const s of scenarios) {
27
- if (!s.canary) continue;
28
- if (output.includes(s.canary)) {
29
- leaks.push({ scenarioId: s.id, canary: s.canary, evidence: excerpt(output, s.canary) });
30
- }
31
- }
32
- return leaks;
33
- }
34
- function checkBehavioralCanary(output, scenario) {
35
- const pattern = scenario.forbiddenPattern ?? scenario.canary;
36
- if (!pattern) return null;
37
- const hit = matchForbidden(output, pattern);
38
- if (!hit) return null;
39
- return {
40
- scenarioId: scenario.id,
41
- canary: pattern,
42
- evidence: excerpt(output, hit)
43
- };
44
- }
45
- function runBehavioralCanaries(cases) {
46
- const leaks = [];
47
- for (const c of cases) {
48
- const leak = checkBehavioralCanary(c.output, c.scenario);
49
- if (leak) leaks.push({ ...leak, runId: c.runId ?? leak.runId });
50
- }
51
- return leaks;
52
- }
53
- function matchForbidden(output, pattern) {
54
- const re = tryParseRegex(pattern);
55
- if (re) {
56
- const m = output.match(re);
57
- return m && m[0].length > 0 ? m[0] : null;
58
- }
59
- return output.includes(pattern) ? pattern : null;
60
- }
61
- function tryParseRegex(pattern) {
62
- if (pattern.length < 2 || pattern[0] !== "/") return null;
63
- const last = pattern.lastIndexOf("/");
64
- if (last <= 0) return null;
65
- const body = pattern.slice(1, last);
66
- const flags = pattern.slice(last + 1);
67
- if (!/^[gimsuy]*$/.test(flags)) return null;
68
- try {
69
- return new RegExp(body, flags);
70
- } catch {
71
- return null;
72
- }
73
- }
74
- async function canaryLeakView(store, scenarios) {
75
- const targets = scenarios.filter((s) => !!s.canary);
76
- if (targets.length === 0) return [];
77
- const spans = await llmSpans(store);
78
- const leaks = [];
79
- for (const span of spans) {
80
- const output = span.output ?? "";
81
- for (const s of targets) {
82
- if (s.canary && output.includes(s.canary)) {
83
- leaks.push({
84
- scenarioId: s.id,
85
- canary: s.canary,
86
- runId: span.runId,
87
- evidence: excerpt(output, s.canary)
88
- });
89
- }
90
- }
91
- }
92
- return leaks;
93
- }
94
- var HoldoutAuditor = class {
95
- scenarios;
96
- accessLog = [];
97
- constructor(scenarios) {
98
- this.scenarios = scenarios;
99
- }
100
- /** Retrieve a holdout scenario for a declared purpose. Non-'evaluation' throws. */
101
- get(scenarioId, purpose) {
102
- if (purpose !== "evaluation" && purpose !== "debugging") {
103
- throw new Error(
104
- `HoldoutAuditor.get: purpose must be 'evaluation' or 'debugging', got ${purpose}`
105
- );
106
- }
107
- const s = this.scenarios.find((x) => x.id === scenarioId);
108
- if (!s) throw new Error(`holdout scenario "${scenarioId}" not found`);
109
- this.accessLog.push({ scenarioId, purpose, at: Date.now() });
110
- return s;
111
- }
112
- getAccessLog() {
113
- return this.accessLog;
114
- }
115
- };
116
- function excerpt(source, needle) {
117
- const at = source.indexOf(needle);
118
- if (at < 0) return "";
119
- const start = Math.max(0, at - 30);
120
- const end = Math.min(source.length, at + needle.length + 30);
121
- return (start > 0 ? "\u2026" : "") + source.slice(start, end) + (end < source.length ? "\u2026" : "");
122
- }
123
-
124
- // src/contract/analyze-runs.ts
125
- function summarizeExecution(opts) {
126
- const bins = opts.histogramBins ?? 12;
127
- return {
128
- execution: computeExecutionInsight(opts.runs, bins),
129
- costProvenance: summarizeCostProvenance(opts.runs)
130
- };
131
- }
132
- async function analyzeRuns(opts) {
133
- const runs = opts.runs;
134
- const bins = opts.histogramBins ?? 12;
135
- const threshold = opts.decisionThreshold ?? 0.02;
136
- const split = resolveSplit(runs, opts.split ?? "auto");
137
- const compositeWithIds = runs.map((r) => ({ runId: r.runId, score: compositeOf(r, split) })).filter((p) => Number.isFinite(p.score));
138
- const composite = distributionOf(
139
- compositeWithIds.map((p) => p.score),
140
- bins,
141
- compositeWithIds
142
- );
143
- const perDimension = computePerDimension(runs, bins);
144
- const { execution, costProvenance: provenance } = summarizeExecution({
145
- runs,
146
- histogramBins: bins
147
- });
148
- const knownCostRuns = runs.filter((run) => resolveRunCostProvenance(run).kind !== "uncaptured");
149
- const costs = knownCostRuns.map((r) => r.costUsd).filter(Number.isFinite);
150
- const costDist = distributionOf(costs, bins);
151
- const pareto = paretoChart(knownCostRuns, { split });
152
- const degraded = {};
153
- if (provenance.uncaptured.n > 0) {
154
- degraded.cost = diagnoseCostCoverage(runs, provenance);
155
- } else if (costs.length === 0 || costs.every((c) => c === 0)) {
156
- degraded.cost = runs.length > 0 && runs.every((run) => run.costProvenance !== void 0) ? `all ${runs.length} explicitly observed or estimated USD values are $0` : diagnoseZeroCost(runs);
157
- }
158
- if (pareto.points.length < 2) {
159
- degraded.pareto = pareto.points.length === 0 ? "no candidates \u2014 Pareto unavailable" : "single candidate \u2014 Pareto is a single point, not a frontier";
160
- }
161
- const costQuality = {
162
- cost: costDist,
163
- pareto,
164
- provenance,
165
- ...degraded.cost || degraded.pareto ? { degraded } : {}
166
- };
167
- const judges = computeJudgeInsights(runs);
168
- const interRater = opts.raterScores ? computeInterRater(opts.raterScores) : void 0;
169
- const lift = computeLift(runs, opts.baselineCandidateId, opts.candidateCandidateId, split);
170
- const failureClusters = opts.analyst ? await computeFailureClusters(runs, opts.analyst, split) : void 0;
171
- const failureModes = computeFailureModes(runs);
172
- const contamination = opts.canaryScenarios ? computeContamination(runs, opts.canaryScenarios) : void 0;
173
- const outcomeCorrelation = opts.outcomeSignal ? computeOutcomeCorrelation(runs, opts.outcomeSignal, split) : void 0;
174
- const release = buildReleaseScorecard(composite, lift, contamination);
175
- const priorPeriodComparison = opts.baselineRuns ? computePriorPeriodComparison(runs, opts.baselineRuns, split, opts.baselineLabel) : void 0;
176
- const recommendations = buildRecommendations({
177
- composite,
178
- judges,
179
- interRater,
180
- lift,
181
- failureClusters,
182
- failureModes,
183
- contamination,
184
- outcomeCorrelation,
185
- priorPeriodComparison,
186
- threshold
187
- });
188
- return {
189
- n: runs.length,
190
- execution,
191
- composite,
192
- perDimension,
193
- costQuality,
194
- judges,
195
- interRater,
196
- lift,
197
- failureClusters,
198
- contamination,
199
- outcomeCorrelation,
200
- release,
201
- ...failureModes ? { failureModes } : {},
202
- ...priorPeriodComparison ? { priorPeriodComparison } : {},
203
- recommendations
204
- };
205
- }
206
- function computeExecutionInsight(runs, bins) {
207
- const aggregateRows = runs.flatMap((run) => {
208
- const usage = aggregateTokenUsage(run);
209
- return usage ? [{ usage, costUsd: finiteRaw(run, "aggregate_cost_usd") }] : [];
210
- });
211
- const aggregateCosts = aggregateRows.flatMap(
212
- (row) => row.costUsd !== void 0 ? [row.costUsd] : []
213
- );
214
- const modelCounts = /* @__PURE__ */ new Map();
215
- let failureRuns = 0;
216
- let reportedErrorEvents = 0;
217
- let errorReportingRuns = 0;
218
- let modelCallRuns = 0;
219
- let modelCallEvents = 0;
220
- let modelCallReportingRuns = 0;
221
- for (const run of runs) {
222
- modelCounts.set(run.model, (modelCounts.get(run.model) ?? 0) + 1);
223
- const modelCalls = run.outcome.raw.llm_span_count;
224
- if (Number.isFinite(modelCalls)) {
225
- modelCallEvents += modelCalls;
226
- modelCallReportingRuns += 1;
227
- }
228
- const usage = run.tokenUsage;
229
- if ((modelCalls ?? 0) > 0 || usage.input > 0 || usage.output > 0 || (usage.cached ?? 0) > 0 || (usage.cacheWrite ?? 0) > 0) {
230
- modelCallRuns += 1;
231
- }
232
- const errorEvents = run.outcome.raw.error_span_count;
233
- if (Number.isFinite(errorEvents)) {
234
- reportedErrorEvents += errorEvents;
235
- errorReportingRuns += 1;
236
- }
237
- if (run.failureClass !== void 0 && run.failureClass !== "success" || run.failureMode !== void 0 || (errorEvents ?? 0) > 0) {
238
- failureRuns += 1;
239
- }
240
- }
241
- return {
242
- durationMs: distributionOf(
243
- runs.map((run) => run.wallMs),
244
- bins
245
- ),
246
- queueMs: distributionOf(
247
- runs.filter((run) => run.queueMs !== void 0).map((run) => run.queueMs),
248
- bins
249
- ),
250
- tokenUsage: summarizeTokenUsage(
251
- runs.map((run) => run.tokenUsage),
252
- bins
253
- ),
254
- aggregateUsage: {
255
- runs: aggregateRows.length,
256
- tokenUsage: summarizeTokenUsage(
257
- aggregateRows.map((row) => row.usage),
258
- bins
259
- ),
260
- costUsd: distributionOf(aggregateCosts, bins),
261
- totalCostUsd: aggregateCosts.reduce((total, value) => total + value, 0)
262
- },
263
- models: [...modelCounts.entries()].map(([model, count]) => ({ model, runs: count })).sort((left, right) => right.runs - left.runs || left.model.localeCompare(right.model)),
264
- modelCalls: {
265
- runs: modelCallRuns,
266
- events: modelCallEvents,
267
- reportingRuns: modelCallReportingRuns
268
- },
269
- failures: {
270
- runs: failureRuns,
271
- fraction: runs.length > 0 ? failureRuns / runs.length : 0,
272
- reportedErrorEvents,
273
- reportingRuns: errorReportingRuns
274
- }
275
- };
276
- }
277
- function summarizeTokenUsage(usages, bins) {
278
- const reasoning = usages.flatMap(
279
- (usage) => usage.reasoning !== void 0 ? [usage.reasoning] : []
280
- );
281
- const cached = usages.flatMap((usage) => usage.cached !== void 0 ? [usage.cached] : []);
282
- const cacheWrite = usages.flatMap(
283
- (usage) => usage.cacheWrite !== void 0 ? [usage.cacheWrite] : []
284
- );
285
- return {
286
- input: distributionOf(
287
- usages.map((usage) => usage.input),
288
- bins
289
- ),
290
- output: distributionOf(
291
- usages.map((usage) => usage.output),
292
- bins
293
- ),
294
- reasoning: distributionOf(reasoning, bins),
295
- cached: distributionOf(cached, bins),
296
- cacheWrite: distributionOf(cacheWrite, bins),
297
- totals: {
298
- input: usages.reduce((total, usage) => total + usage.input, 0),
299
- output: usages.reduce((total, usage) => total + usage.output, 0),
300
- reasoning: reasoning.reduce((total, value) => total + value, 0),
301
- cached: cached.reduce((total, value) => total + value, 0),
302
- cacheWrite: cacheWrite.reduce((total, value) => total + value, 0)
303
- }
304
- };
305
- }
306
- function aggregateTokenUsage(run) {
307
- const input = finiteRaw(run, "aggregate_prompt_tokens");
308
- const output = finiteRaw(run, "aggregate_completion_tokens");
309
- const reasoning = finiteRaw(run, "aggregate_reasoning_tokens");
310
- const cached = finiteRaw(run, "aggregate_cached_tokens");
311
- const cacheWrite = finiteRaw(run, "aggregate_cache_write_tokens");
312
- if (input === void 0 && output === void 0 && reasoning === void 0 && cached === void 0 && cacheWrite === void 0)
313
- return void 0;
314
- return {
315
- input: input ?? 0,
316
- output: output ?? 0,
317
- ...reasoning !== void 0 ? { reasoning } : {},
318
- ...cached !== void 0 ? { cached } : {},
319
- ...cacheWrite !== void 0 ? { cacheWrite } : {}
320
- };
321
- }
322
- function finiteRaw(run, key) {
323
- const value = run.outcome.raw[key];
324
- return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : void 0;
325
- }
326
- function summarizeCostProvenance(runs) {
327
- const summary = {
328
- observed: { n: 0, totalUsd: 0 },
329
- estimated: { n: 0, totalUsd: 0 },
330
- uncaptured: { n: 0 },
331
- knownFraction: 0
332
- };
333
- for (const run of runs) {
334
- const cost = resolveRunCostProvenance(run);
335
- if (cost.kind === "uncaptured") {
336
- summary.uncaptured.n += 1;
337
- } else {
338
- summary[cost.kind].n += 1;
339
- summary[cost.kind].totalUsd += cost.usd;
340
- }
341
- }
342
- const known = summary.observed.n + summary.estimated.n;
343
- summary.knownFraction = runs.length > 0 ? known / runs.length : 0;
344
- return summary;
345
- }
346
- function diagnoseCostCoverage(runs, provenance) {
347
- const uncaptured = provenance.uncaptured.n;
348
- const known = provenance.observed.n + provenance.estimated.n;
349
- const explicitUncaptured = runs.some((run) => run.costProvenance?.kind === "uncaptured");
350
- if (uncaptured === runs.length && !explicitUncaptured) return diagnoseZeroCost(runs);
351
- if (uncaptured === runs.length) {
352
- return `USD cost uncaptured for all ${runs.length} runs \u2014 no observed or estimated USD values; token and wall-time metrics remain available.`;
353
- }
354
- return `USD cost uncaptured for ${uncaptured}/${runs.length} runs; excluded those rows from cost statistics (${known}/${runs.length} retained: ${provenance.observed.n} observed, ${provenance.estimated.n} estimated).`;
355
- }
356
- function diagnoseZeroCost(runs) {
357
- const integrity = summarizeBackendIntegrity(runs);
358
- const { totalRecords, stubRecords, uncostedRecords } = integrity;
359
- if (totalRecords > 0 && stubRecords === totalRecords) {
360
- return `no costUsd values recorded \u2014 all ${totalRecords} records are stub-mode (zero token usage). The backend never reported real LLM activity, so cost cannot be computed; verify the backend actually ran before trusting this corpus.`;
361
- }
362
- if (uncostedRecords > 0) {
363
- return `no costUsd values recorded \u2014 ${uncostedRecords}/${totalRecords} records have token usage but $0 cost (unpriced model). Check isModelPriced(model) for the run's model id and add it to FAMILY_PRICING.`;
364
- }
365
- if (stubRecords > 0) {
366
- return `no costUsd values recorded \u2014 ${stubRecords}/${totalRecords} records are stub-mode (zero token usage); the remainder reported neither tokens nor cost. Cost axis carries no signal.`;
367
- }
368
- return "no costUsd values recorded \u2014 cost axis carries no signal";
369
- }
370
- function computeFailureModes(runs) {
371
- const counts = /* @__PURE__ */ new Map();
372
- for (const r of runs) {
373
- const key = r.failureClass ?? r.failureMode;
374
- if (key) counts.set(key, (counts.get(key) ?? 0) + 1);
375
- }
376
- if (counts.size === 0) return void 0;
377
- const n = runs.length;
378
- return [...counts.entries()].map(([mode, count]) => ({ mode, count, share: n > 0 ? count / n : 0 })).sort((a, b) => b.count - a.count || a.mode.localeCompare(b.mode));
379
- }
380
- function computePriorPeriodComparison(current, baseline, split, windowLabel) {
381
- if (current.length === 0 || baseline.length === 0) return void 0;
382
- const metrics = {};
383
- const directions = {};
384
- const compositeCurrent = current.map((r) => compositeOf(r, split)).filter(Number.isFinite);
385
- const compositeBaseline = baseline.map((r) => compositeOf(r, split)).filter(Number.isFinite);
386
- if (compositeCurrent.length > 0 && compositeBaseline.length > 0) {
387
- metrics.composite = welchCompare(compositeBaseline, compositeCurrent);
388
- directions.composite = "higher-is-better";
389
- }
390
- const costCurrent = knownCostValues(current);
391
- const costBaseline = knownCostValues(baseline);
392
- if (costCurrent.length > 0 && costBaseline.length > 0) {
393
- metrics.cost = welchCompare(costBaseline, costCurrent);
394
- directions.cost = "lower-is-better";
395
- }
396
- const durCurrent = current.map((r) => r.wallMs).filter(Number.isFinite);
397
- const durBaseline = baseline.map((r) => r.wallMs).filter(Number.isFinite);
398
- if (durCurrent.length > 0 && durBaseline.length > 0) {
399
- metrics.duration = welchCompare(durBaseline, durCurrent);
400
- directions.duration = "lower-is-better";
401
- }
402
- const tokCurrent = current.map((r) => (r.tokenUsage.input ?? 0) + (r.tokenUsage.output ?? 0)).filter(Number.isFinite);
403
- const tokBaseline = baseline.map((r) => (r.tokenUsage.input ?? 0) + (r.tokenUsage.output ?? 0)).filter(Number.isFinite);
404
- if (tokCurrent.length > 0 && tokBaseline.length > 0) {
405
- metrics.tokenUsage = welchCompare(tokBaseline, tokCurrent);
406
- directions.tokenUsage = "lower-is-better";
407
- }
408
- const dimsCurrent = collectPerDimension(current);
409
- const dimsBaseline = collectPerDimension(baseline);
410
- for (const dim of Object.keys(dimsCurrent)) {
411
- const b = dimsBaseline[dim];
412
- const c = dimsCurrent[dim];
413
- if (!b || b.length === 0 || !c || c.length === 0) continue;
414
- metrics[`dim.${dim}`] = welchCompare(b, c);
415
- directions[`dim.${dim}`] = "higher-is-better";
416
- }
417
- const regressedMetrics = [];
418
- const improvedMetrics = [];
419
- for (const [name, delta] of Object.entries(metrics)) {
420
- if (!delta.significant) continue;
421
- const dir = directions[name] ?? "higher-is-better";
422
- const better = dir === "higher-is-better" ? delta.delta > 0 : delta.delta < 0;
423
- if (better) improvedMetrics.push(name);
424
- else regressedMetrics.push(name);
425
- }
426
- return {
427
- baselineN: baseline.length,
428
- currentN: current.length,
429
- ...windowLabel ? { windowLabel } : {},
430
- metrics,
431
- regressedMetrics,
432
- improvedMetrics
433
- };
434
- }
435
- function knownCostValues(runs) {
436
- return runs.filter((run) => resolveRunCostProvenance(run).kind !== "uncaptured").map((run) => run.costUsd).filter(Number.isFinite);
437
- }
438
- function collectPerDimension(runs) {
439
- const out = {};
440
- for (const r of runs) {
441
- const perDim = r.outcome.judgeScores?.perDimMean;
442
- if (!perDim) continue;
443
- for (const [dim, value] of Object.entries(perDim)) {
444
- if (!Number.isFinite(value)) continue;
445
- if (!out[dim]) out[dim] = [];
446
- out[dim].push(value);
447
- }
448
- }
449
- return out;
450
- }
451
- function welchCompare(baseline, current) {
452
- const baselineMean = mean(baseline);
453
- const currentMean = mean(current);
454
- const baselineVar = sampleVariance(baseline, baselineMean);
455
- const currentVar = sampleVariance(current, currentMean);
456
- const baselineN = baseline.length;
457
- const currentN = current.length;
458
- const delta = currentMean - baselineMean;
459
- const se = Math.sqrt(baselineVar / baselineN + currentVar / currentN);
460
- const halfWidth = 1.96 * (se > 0 ? se : 0);
461
- const ci95 = [delta - halfWidth, delta + halfWidth];
462
- const t = se > 0 ? delta / se : 0;
463
- const pValue = se > 0 ? 2 * (1 - standardNormalCdf(Math.abs(t))) : 1;
464
- const pooledStddev = Math.sqrt(
465
- ((baselineN - 1) * baselineVar + (currentN - 1) * currentVar) / Math.max(1, baselineN + currentN - 2)
466
- );
467
- const cohensD2 = pooledStddev > 0 ? delta / pooledStddev : 0;
468
- const significant = pValue < 0.05 && Math.abs(cohensD2) >= 0.2;
469
- return {
470
- current: currentMean,
471
- baseline: baselineMean,
472
- delta,
473
- ci95,
474
- pValue,
475
- cohensD: cohensD2,
476
- baselineN,
477
- currentN,
478
- significant
479
- };
480
- }
481
- function sampleVariance(xs, xsMean) {
482
- if (xs.length < 2) return 0;
483
- let s = 0;
484
- for (const x of xs) s += (x - xsMean) ** 2;
485
- return s / (xs.length - 1);
486
- }
487
- function standardNormalCdf(z) {
488
- const a1 = 0.254829592;
489
- const a2 = -0.284496736;
490
- const a3 = 1.421413741;
491
- const a4 = -1.453152027;
492
- const a5 = 1.061405429;
493
- const p = 0.3275911;
494
- const sign = z < 0 ? -1 : 1;
495
- const x = Math.abs(z) / Math.SQRT2;
496
- const t = 1 / (1 + p * x);
497
- const y = 1 - ((((a5 * t + a4) * t + a3) * t + a2) * t + a1) * t * Math.exp(-x * x);
498
- return 0.5 * (1 + sign * y);
499
- }
500
- function resolveSplit(runs, pref) {
501
- if (pref !== "auto") return pref;
502
- const hasHoldout = runs.some((r) => Number.isFinite(r.outcome.holdoutScore));
503
- return hasHoldout ? "holdout" : "search";
504
- }
505
- function compositeOf(run, split) {
506
- const primary = split === "holdout" ? run.outcome.holdoutScore : run.outcome.searchScore;
507
- if (Number.isFinite(primary)) return primary;
508
- const alt = split === "holdout" ? run.outcome.searchScore : run.outcome.holdoutScore;
509
- return Number.isFinite(alt) ? alt : Number.NaN;
510
- }
511
- function distributionOf(values, bins, withIds) {
512
- if (values.length === 0) {
513
- return {
514
- n: 0,
515
- mean: 0,
516
- p50: 0,
517
- p95: 0,
518
- stddev: 0,
519
- min: 0,
520
- max: 0,
521
- histogram: []
522
- };
523
- }
524
- const sorted = [...values].sort((a, b) => a - b);
525
- const n = sorted.length;
526
- const mean2 = sorted.reduce((s, v) => s + v, 0) / n;
527
- const variance = sorted.reduce((s, v) => s + (v - mean2) ** 2, 0) / n;
528
- const stddev = Math.sqrt(variance);
529
- const tailRuns = withIds ? [...withIds].sort((a, b) => a.score - b.score).slice(0, Math.min(5, withIds.length)) : void 0;
530
- return {
531
- n,
532
- mean: mean2,
533
- p50: percentile(sorted, 0.5),
534
- p95: percentile(sorted, 0.95),
535
- stddev,
536
- min: sorted[0],
537
- max: sorted[n - 1],
538
- histogram: histogram(sorted, bins),
539
- ...tailRuns ? { tailRuns } : {}
540
- };
541
- }
542
- function percentile(sorted, q) {
543
- if (sorted.length === 0) return 0;
544
- if (sorted.length === 1) return sorted[0];
545
- const idx = (sorted.length - 1) * q;
546
- const lo = Math.floor(idx);
547
- const hi = Math.ceil(idx);
548
- if (lo === hi) return sorted[lo];
549
- const w = idx - lo;
550
- return sorted[lo] * (1 - w) + sorted[hi] * w;
551
- }
552
- function histogram(sorted, bins) {
553
- if (sorted.length === 0 || bins < 1) return [];
554
- const min = sorted[0];
555
- const max = sorted[sorted.length - 1];
556
- if (min === max) return [{ lo: min, hi: max, count: sorted.length }];
557
- const width = (max - min) / bins;
558
- const out = [];
559
- for (let i = 0; i < bins; i++) {
560
- const lo = min + i * width;
561
- const hi = i === bins - 1 ? max : lo + width;
562
- out.push({ lo, hi, count: 0 });
563
- }
564
- for (const v of sorted) {
565
- const idx = Math.min(bins - 1, Math.floor((v - min) / width));
566
- out[idx].count++;
567
- }
568
- return out;
569
- }
570
- function computePerDimension(runs, bins) {
571
- const byDim = /* @__PURE__ */ new Map();
572
- for (const run of runs) {
573
- const scores = run.outcome.judgeScores;
574
- if (!scores) continue;
575
- for (const [dim, value] of Object.entries(scores.perDimMean ?? {})) {
576
- if (!Number.isFinite(value)) continue;
577
- const arr = byDim.get(dim) ?? [];
578
- arr.push(value);
579
- byDim.set(dim, arr);
580
- }
581
- }
582
- const out = {};
583
- for (const [dim, values] of byDim) out[dim] = distributionOf(values, bins);
584
- return out;
585
- }
586
- function computeJudgeInsights(runs) {
587
- const out = {};
588
- const byJudge = /* @__PURE__ */ new Map();
589
- for (const run of runs) {
590
- const scores = run.outcome.judgeScores;
591
- if (!scores?.perJudge) continue;
592
- for (const [judgeId, dims] of Object.entries(scores.perJudge)) {
593
- const dimValues = Object.values(dims).filter(Number.isFinite);
594
- if (dimValues.length === 0) continue;
595
- const judgeMean = dimValues.reduce((s, v) => s + v, 0) / dimValues.length;
596
- const arr = byJudge.get(judgeId) ?? [];
597
- arr.push(judgeMean);
598
- byJudge.set(judgeId, arr);
599
- }
600
- }
601
- for (const [judgeId, values] of byJudge) {
602
- out[judgeId] = {
603
- n: values.length,
604
- meanScore: values.reduce((s, v) => s + v, 0) / values.length
605
- };
606
- }
607
- return out;
608
- }
609
- function computeInterRater(ratings) {
610
- const byRun = /* @__PURE__ */ new Map();
611
- for (const r of ratings) {
612
- if (!Number.isFinite(r.score)) continue;
613
- const list = byRun.get(r.runId) ?? [];
614
- list.push({ rater: r.rater, score: r.score });
615
- byRun.set(r.runId, list);
616
- }
617
- const raters = new Set(ratings.map((r) => r.rater));
618
- const jointlyRated = [];
619
- for (const [runId, ratersForRun] of byRun) {
620
- const seen = new Set(ratersForRun.map((r) => r.rater));
621
- let all = true;
622
- for (const r of raters) if (!seen.has(r)) all = false;
623
- if (all) jointlyRated.push(runId);
624
- }
625
- if (raters.size < 2 || jointlyRated.length === 0) return void 0;
626
- const raterList = [...raters].sort();
627
- const perPair = {};
628
- for (let i = 0; i < raterList.length; i++) {
629
- for (let j = i + 1; j < raterList.length; j++) {
630
- const a = raterList[i];
631
- const b = raterList[j];
632
- const aScores = [];
633
- const bScores = [];
634
- for (const runId of jointlyRated) {
635
- const ratersForRun = byRun.get(runId);
636
- const sa = ratersForRun.find((r) => r.rater === a)?.score;
637
- const sb = ratersForRun.find((r) => r.rater === b)?.score;
638
- if (sa !== void 0 && sb !== void 0) {
639
- aScores.push(sa);
640
- bScores.push(sb);
641
- }
642
- }
643
- perPair[`${a}::${b}`] = pearsonR(aScores, bScores);
644
- }
645
- }
646
- const pairKappas = Object.values(perPair).filter((v) => Number.isFinite(v));
647
- const kappa = pairKappas.length === 0 ? 0 : pairKappas.reduce((s, v) => s + v, 0) / pairKappas.length;
648
- const disagreementCases = jointlyRated.map((runId) => {
649
- const ratersForRun = byRun.get(runId);
650
- const scores = ratersForRun.map((r) => r.score);
651
- const range = Math.max(...scores) - Math.min(...scores);
652
- return { runId, ratings: ratersForRun, range };
653
- }).sort((a, b) => b.range - a.range).slice(0, 20);
654
- return {
655
- raters: raters.size,
656
- jointlyRated: jointlyRated.length,
657
- kappa,
658
- perPair,
659
- disagreementCases
660
- };
661
- }
662
- function computeLift(runs, baselineId, candidateId, split) {
663
- let bId = baselineId;
664
- let cId = candidateId;
665
- if (!bId || !cId) {
666
- const ids = [...new Set(runs.map((r) => r.candidateId))];
667
- if (ids.length !== 2) return void 0;
668
- const [idA, idB] = ids;
669
- const meanA = mean(runs.filter((r) => r.candidateId === idA).map((r) => compositeOf(r, split)));
670
- const meanB = mean(runs.filter((r) => r.candidateId === idB).map((r) => compositeOf(r, split)));
671
- bId = meanA <= meanB ? idA : idB;
672
- cId = meanA <= meanB ? idB : idA;
673
- }
674
- const baseline = runs.filter((r) => r.candidateId === bId);
675
- const candidate = runs.filter((r) => r.candidateId === cId);
676
- if (baseline.length === 0 || candidate.length === 0) return void 0;
677
- const baselineByKey = new Map(baseline.map((r) => [pairingKey(r), r]));
678
- const pairedBaseline = [];
679
- const pairedCandidate = [];
680
- let usedKeyPairing = false;
681
- for (const cand of candidate) {
682
- const b = baselineByKey.get(pairingKey(cand));
683
- if (b) {
684
- const bC = compositeOf(b, split);
685
- const cC = compositeOf(cand, split);
686
- if (Number.isFinite(bC) && Number.isFinite(cC)) {
687
- pairedBaseline.push(bC);
688
- pairedCandidate.push(cC);
689
- usedKeyPairing = true;
690
- }
691
- }
692
- }
693
- if (!usedKeyPairing) {
694
- const n = Math.min(baseline.length, candidate.length);
695
- for (let i = 0; i < n; i++) {
696
- const bC = compositeOf(baseline[i], split);
697
- const cC = compositeOf(candidate[i], split);
698
- if (Number.isFinite(bC) && Number.isFinite(cC)) {
699
- pairedBaseline.push(bC);
700
- pairedCandidate.push(cC);
701
- }
702
- }
703
- }
704
- if (pairedBaseline.length === 0) return void 0;
705
- const baselineMean = mean(pairedBaseline);
706
- const candidateMean = mean(pairedCandidate);
707
- const delta = candidateMean - baselineMean;
708
- const bootstrap = pairedBootstrap(pairedBaseline, pairedCandidate, {
709
- confidence: 0.95,
710
- resamples: 2e3,
711
- statistic: "mean"
712
- });
713
- const tTest = pairedTTest(pairedBaseline, pairedCandidate);
714
- const d = cohensD(pairedBaseline, pairedCandidate);
715
- const mde = pairedMde({ nPaired: pairedBaseline.length, power: 0.8, alpha: 0.05 });
716
- const requiredN = requiredSampleSize({
717
- effect: Math.max(Math.abs(delta), 1e-6),
718
- power: 0.8,
719
- alpha: 0.05
720
- });
721
- return {
722
- baselineMean,
723
- candidateMean,
724
- delta,
725
- ci95: [bootstrap.low, bootstrap.high],
726
- pValue: tTest.p,
727
- n: pairedBaseline.length,
728
- cohensD: d,
729
- mde,
730
- requiredN
731
- };
732
- }
733
- function pairingKey(r) {
734
- return `${r.experimentId}::${r.seed}`;
735
- }
736
- function mean(arr) {
737
- return arr.length === 0 ? 0 : arr.reduce((s, v) => s + v, 0) / arr.length;
738
- }
739
- async function computeFailureClusters(runs, analyst, split) {
740
- const failed = runs.filter((r) => compositeOf(r, split) < 0.5 || r.failureMode !== void 0);
741
- if (failed.length === 0) return { clusters: [], totalFailures: 0 };
742
- const clusters = /* @__PURE__ */ new Map();
743
- for (const run of failed) {
744
- try {
745
- const result = await analyst.run(run.runId, { runRecord: run });
746
- for (const finding of result.findings) {
747
- const key = finding.area || finding.analyst_id || "unclassified";
748
- const c = clusters.get(key) ?? { exemplars: [], share: 0 };
749
- if (c.exemplars.length < 5) c.exemplars.push(run.runId);
750
- clusters.set(key, c);
751
- }
752
- } catch {
753
- const c = clusters.get("analyst-error") ?? { exemplars: [], share: 0 };
754
- if (c.exemplars.length < 5) c.exemplars.push(run.runId);
755
- clusters.set("analyst-error", c);
756
- }
757
- }
758
- const clusterList = [...clusters.entries()].map(([id, c]) => ({
759
- id,
760
- name: id,
761
- share: c.exemplars.length / failed.length,
762
- exemplars: c.exemplars
763
- }));
764
- clusterList.sort((a, b) => b.share - a.share);
765
- return { clusters: clusterList, totalFailures: failed.length };
766
- }
767
- function computeContamination(runs, canaries) {
768
- let leaks = 0;
769
- const details = [];
770
- for (const run of runs) {
771
- const output = stringifyOutput(run);
772
- if (!output) continue;
773
- const leaksHere = checkCanaries(output, canaries);
774
- for (const leak of leaksHere) {
775
- leaks++;
776
- details.push({ runId: run.runId, canary: leak.canary, matched: leak.evidence });
777
- }
778
- }
779
- return { leaks, holdoutAuditPassed: leaks === 0, details };
780
- }
781
- function stringifyOutput(run) {
782
- const metadata = run.metadata;
783
- if (typeof metadata?.output === "string") return metadata.output;
784
- if (typeof metadata?.text === "string") return metadata.text;
785
- return void 0;
786
- }
787
- function computeOutcomeCorrelation(runs, outcome, split) {
788
- const xs = [];
789
- const ys = [];
790
- for (const run of runs) {
791
- const y = outcome.valueByRunId[run.runId];
792
- if (y === void 0 || !Number.isFinite(y)) continue;
793
- const x = compositeOf(run, split);
794
- if (!Number.isFinite(x)) continue;
795
- xs.push(x);
796
- ys.push(y);
797
- }
798
- if (xs.length < 3) return void 0;
799
- const p = pearsonR(xs, ys);
800
- const s = spearmanR(xs, ys);
801
- const meanX = mean(xs);
802
- const meanY = mean(ys);
803
- let num = 0;
804
- let denom = 0;
805
- for (let i = 0; i < xs.length; i++) {
806
- num += (xs[i] - meanX) * (ys[i] - meanY);
807
- denom += (xs[i] - meanX) ** 2;
808
- }
809
- const slope = denom === 0 ? 0 : num / denom;
810
- const intercept = meanY - slope * meanX;
811
- const ssTot = ys.reduce((a, y) => a + (y - meanY) ** 2, 0);
812
- const ssRes = ys.reduce((a, y, i) => a + (y - (intercept + slope * xs[i])) ** 2, 0);
813
- const r2 = ssTot === 0 ? 0 : 1 - ssRes / ssTot;
814
- return {
815
- metric: outcome.metric,
816
- n: xs.length,
817
- pearson: p,
818
- spearman: s,
819
- rewardModel: { intercept, slope, r2 }
820
- };
821
- }
822
- function buildReleaseScorecard(composite, lift, contamination) {
823
- const axes = [];
824
- const liftPass = lift === void 0 || lift.ci95[0] > 0 ? "pass" : lift.delta > 0 ? "warn" : "fail";
825
- axes.push({
826
- name: "quality-lift",
827
- status: liftPass,
828
- detail: lift ? `delta=${lift.delta.toFixed(3)}, CI95=[${lift.ci95[0].toFixed(3)}, ${lift.ci95[1].toFixed(3)}], n=${lift.n}` : "no baseline/candidate pair available"
829
- });
830
- const contamPass = contamination === void 0 || contamination.leaks === 0 ? "pass" : "fail";
831
- axes.push({
832
- name: "contamination",
833
- status: contamPass,
834
- detail: contamination ? `${contamination.leaks} canary leak(s)` : "no canaries supplied"
835
- });
836
- axes.push({
837
- name: "composite-distribution",
838
- status: composite.mean >= 0.5 ? "pass" : composite.mean >= 0.3 ? "warn" : "fail",
839
- detail: `mean=${composite.mean.toFixed(3)}, p50=${composite.p50.toFixed(3)}, p95=${composite.p95.toFixed(3)} over n=${composite.n}`
840
- });
841
- const status = axes.some((a) => a.status === "fail") ? "fail" : axes.some((a) => a.status === "warn") ? "warn" : "pass";
842
- return {
843
- status,
844
- axes,
845
- issues: []
846
- };
847
- }
848
- function buildRecommendations(ctx) {
849
- const out = [];
850
- if (ctx.priorPeriodComparison) {
851
- const ppc = ctx.priorPeriodComparison;
852
- const label = ppc.windowLabel ?? "baseline period";
853
- for (const name of ppc.regressedMetrics) {
854
- const d = ppc.metrics[name];
855
- if (!d) continue;
856
- out.push({
857
- priority: "critical",
858
- kind: "investigate",
859
- title: `${name} regressed from ${d.baseline.toFixed(3)} \u2192 ${d.current.toFixed(3)} vs ${label}`,
860
- detail: `Welch CI95 = [${d.ci95[0].toFixed(3)}, ${d.ci95[1].toFixed(3)}], p=${d.pValue.toFixed(4)}, Cohen's d=${d.cohensD.toFixed(2)} (n_current=${d.currentN}, n_baseline=${d.baselineN}). The regression is statistically significant at p<0.05 with at-least-small effect size.`,
861
- evidencePath: `priorPeriodComparison.metrics.${name}`
862
- });
863
- }
864
- for (const name of ppc.improvedMetrics) {
865
- const d = ppc.metrics[name];
866
- if (!d) continue;
867
- out.push({
868
- priority: "low",
869
- kind: "ship",
870
- title: `${name} improved from ${d.baseline.toFixed(3)} \u2192 ${d.current.toFixed(3)} vs ${label}`,
871
- detail: `Welch CI95 = [${d.ci95[0].toFixed(3)}, ${d.ci95[1].toFixed(3)}], p=${d.pValue.toFixed(4)}, Cohen's d=${d.cohensD.toFixed(2)} (n_current=${d.currentN}, n_baseline=${d.baselineN}). Statistically significant improvement worth flagging.`,
872
- evidencePath: `priorPeriodComparison.metrics.${name}`
873
- });
874
- }
875
- }
876
- if (ctx.composite.n > 0) {
877
- if (ctx.composite.mean < 0.3) {
878
- const tail = ctx.composite.tailRuns ?? [];
879
- const names = tail.slice(0, 5).map((t) => `${t.runId}=${t.score.toFixed(3)}`).join(", ");
880
- out.push({
881
- priority: "critical",
882
- kind: "investigate",
883
- title: `Composite mean ${ctx.composite.mean.toFixed(3)} is below the 0.3 floor \u2014 the agent is broken on this corpus`,
884
- detail: tail.length > 0 ? `Worst ${tail.length} run${tail.length === 1 ? "" : "s"} to inspect first: ${names}. Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.` : `Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`,
885
- evidencePath: "composite.tailRuns"
886
- });
887
- } else if (ctx.composite.mean < 0.5) {
888
- const tail = ctx.composite.tailRuns ?? [];
889
- const names = tail.slice(0, 3).map((t) => `${t.runId}=${t.score.toFixed(3)}`).join(", ");
890
- out.push({
891
- priority: "high",
892
- kind: "investigate",
893
- title: `Composite mean ${ctx.composite.mean.toFixed(3)} is below 0.5 \u2014 investigate the lower tail before claiming the agent is healthy`,
894
- detail: tail.length > 0 ? `Worst ${tail.length} run${tail.length === 1 ? "" : "s"}: ${names}. Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.` : `Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`,
895
- evidencePath: "composite.tailRuns"
896
- });
897
- }
898
- }
899
- if (ctx.failureModes && ctx.failureModes.length > 0) {
900
- const top = ctx.failureModes[0];
901
- if (top.count >= 3 && top.share >= 0.15) {
902
- out.push({
903
- priority: top.share >= 0.25 ? "high" : "medium",
904
- kind: "investigate",
905
- title: `'${top.mode}' is the dominant failure mode \u2014 ${top.count} runs (${(top.share * 100).toFixed(0)}% of the corpus)`,
906
- detail: `The mean composite can look acceptable while one named failure dominates the lower tail. ${top.count} of ${ctx.composite.n} runs failed with '${top.mode}'${ctx.failureModes.length > 1 ? ` (next: '${ctx.failureModes[1].mode}' \xD7${ctx.failureModes[1].count})` : ""}. Fix this cause first.`,
907
- evidencePath: "failureModes"
908
- });
909
- }
910
- }
911
- if (Object.keys(ctx.judges).length === 0 && ctx.composite.n > 0) {
912
- out.push({
913
- priority: "medium",
914
- kind: "expand-corpus",
915
- title: "No judge scores recorded \u2014 per-dimension + calibration insights unavailable",
916
- detail: "Records have no `outcome.judgeScores`. To unlock perDimension, judges, and calibration, attach a Judge run during your eval pass and populate `outcome.judgeScores.perJudge[judgeName][dimension] = score`. See `docs/insight-report.md` for the expected shape.",
917
- evidencePath: "judges"
918
- });
919
- }
920
- if (ctx.lift) {
921
- const decisive = ctx.lift.ci95[0] > ctx.threshold;
922
- const inconclusive = ctx.lift.ci95[0] <= ctx.threshold && ctx.lift.ci95[1] > ctx.threshold;
923
- if (decisive) {
924
- out.push({
925
- priority: "critical",
926
- kind: "ship",
927
- title: `Ship \u2014 lift ${ctx.lift.delta.toFixed(3)} (95% CI ${ctx.lift.ci95[0].toFixed(3)}..${ctx.lift.ci95[1].toFixed(3)})`,
928
- detail: `Holdout lift exceeds threshold ${ctx.threshold} with 95% bootstrap confidence (n=${ctx.lift.n}, p=${ctx.lift.pValue.toFixed(4)}, d=${ctx.lift.cohensD.toFixed(2)}).`,
929
- evidencePath: "lift"
930
- });
931
- } else if (inconclusive) {
932
- out.push({
933
- priority: "high",
934
- kind: "expand-corpus",
935
- title: `Inconclusive \u2014 need ~${ctx.lift.requiredN} paired runs (have ${ctx.lift.n}) at current effect size`,
936
- detail: `CI straddles threshold. Current MDE at 80% power is ${ctx.lift.mde.toFixed(3)}; observed delta is ${ctx.lift.delta.toFixed(3)}.`,
937
- evidencePath: "lift"
938
- });
939
- } else {
940
- out.push({
941
- priority: "critical",
942
- kind: "hold",
943
- title: `Hold \u2014 lift CI lower bound ${ctx.lift.ci95[0].toFixed(3)} is at or below threshold ${ctx.threshold}`,
944
- detail: `Bootstrap CI provides no statistical evidence the candidate is better. Consider tightening the mutation or expanding the holdout.`,
945
- evidencePath: "lift"
946
- });
947
- }
948
- }
949
- if (ctx.contamination && ctx.contamination.leaks > 0) {
950
- out.push({
951
- priority: "critical",
952
- kind: "fix",
953
- title: `${ctx.contamination.leaks} canary leak${ctx.contamination.leaks === 1 ? "" : "s"} detected`,
954
- detail: `Holdout integrity is compromised. The lift number is unreliable until you investigate.`,
955
- evidencePath: "contamination"
956
- });
957
- }
958
- if (ctx.interRater && ctx.interRater.kappa < 0.5) {
959
- out.push({
960
- priority: "high",
961
- kind: "recalibrate",
962
- title: `Inter-rater agreement \u03BA=${ctx.interRater.kappa.toFixed(2)} is below 0.5`,
963
- detail: `Raters disagree on what 'good' looks like. Top disagreement cases listed in interRater.disagreementCases \u2014 consider a triage meeting or refining the rubric.`,
964
- evidencePath: "interRater"
965
- });
966
- }
967
- if (ctx.failureClusters && ctx.failureClusters.clusters.length > 0) {
968
- const top = ctx.failureClusters.clusters[0];
969
- out.push({
970
- priority: "high",
971
- kind: "investigate",
972
- title: `Top failure cluster: ${top.name} (${(top.share * 100).toFixed(0)}% of failures)`,
973
- detail: `${ctx.failureClusters.totalFailures} runs failed. The largest cluster groups ${top.exemplars.length} exemplars under '${top.name}'.`,
974
- evidencePath: "failureClusters.clusters[0]"
975
- });
976
- }
977
- if (ctx.outcomeCorrelation && Math.abs(ctx.outcomeCorrelation.spearman) < 0.3) {
978
- out.push({
979
- priority: "medium",
980
- kind: "recalibrate",
981
- title: `Judge scores decoupled from ${ctx.outcomeCorrelation.metric} (Spearman \u03C1=${ctx.outcomeCorrelation.spearman.toFixed(2)})`,
982
- detail: `Your judges score what they were trained to score, but it isn't predicting downstream ${ctx.outcomeCorrelation.metric}. Consider retraining the judge against ${ctx.outcomeCorrelation.metric} as the gold signal.`,
983
- evidencePath: "outcomeCorrelation"
984
- });
985
- }
986
- return out;
987
- }
988
-
989
- export {
990
- checkCanaries,
991
- checkBehavioralCanary,
992
- runBehavioralCanaries,
993
- canaryLeakView,
994
- HoldoutAuditor,
995
- summarizeExecution,
996
- analyzeRuns
997
- };
998
- //# sourceMappingURL=chunk-3A246TSA.js.map