@tangle-network/agent-eval 0.120.0 → 0.120.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/package.json +1 -1
  3. package/dist/analyst/index.d.ts +0 -3111
  4. package/dist/analyst/index.js +0 -403
  5. package/dist/analyst/index.js.map +0 -1
  6. package/dist/authenticity/index.d.ts +0 -161
  7. package/dist/authenticity/index.js +0 -215
  8. package/dist/authenticity/index.js.map +0 -1
  9. package/dist/belief-state/index.d.ts +0 -1301
  10. package/dist/belief-state/index.js +0 -2152
  11. package/dist/belief-state/index.js.map +0 -1
  12. package/dist/benchmarks/index.d.ts +0 -974
  13. package/dist/benchmarks/index.js +0 -60
  14. package/dist/benchmarks/index.js.map +0 -1
  15. package/dist/builder-eval/index.d.ts +0 -695
  16. package/dist/builder-eval/index.js +0 -366
  17. package/dist/builder-eval/index.js.map +0 -1
  18. package/dist/campaign/index.d.ts +0 -7454
  19. package/dist/campaign/index.js +0 -272
  20. package/dist/campaign/index.js.map +0 -1
  21. package/dist/chunk-32BZXMSO.js +0 -3878
  22. package/dist/chunk-32BZXMSO.js.map +0 -1
  23. package/dist/chunk-3A246TSA.js +0 -998
  24. package/dist/chunk-3A246TSA.js.map +0 -1
  25. package/dist/chunk-3RF76KTD.js +0 -84
  26. package/dist/chunk-3RF76KTD.js.map +0 -1
  27. package/dist/chunk-3YYRZDON.js +0 -45
  28. package/dist/chunk-3YYRZDON.js.map +0 -1
  29. package/dist/chunk-4I2E3LLO.js +0 -1030
  30. package/dist/chunk-4I2E3LLO.js.map +0 -1
  31. package/dist/chunk-ARU2PZFM.js +0 -312
  32. package/dist/chunk-ARU2PZFM.js.map +0 -1
  33. package/dist/chunk-BOD4O7OF.js +0 -40
  34. package/dist/chunk-BOD4O7OF.js.map +0 -1
  35. package/dist/chunk-DPZAEKA6.js +0 -880
  36. package/dist/chunk-DPZAEKA6.js.map +0 -1
  37. package/dist/chunk-DTJ6QUQB.js +0 -131
  38. package/dist/chunk-DTJ6QUQB.js.map +0 -1
  39. package/dist/chunk-GGE4NNQT.js +0 -65
  40. package/dist/chunk-GGE4NNQT.js.map +0 -1
  41. package/dist/chunk-H5UD2323.js +0 -286
  42. package/dist/chunk-H5UD2323.js.map +0 -1
  43. package/dist/chunk-HHWE3POT.js +0 -94
  44. package/dist/chunk-HHWE3POT.js.map +0 -1
  45. package/dist/chunk-HKUCJ437.js +0 -787
  46. package/dist/chunk-HKUCJ437.js.map +0 -1
  47. package/dist/chunk-JHCHEVET.js +0 -274
  48. package/dist/chunk-JHCHEVET.js.map +0 -1
  49. package/dist/chunk-JHOJHHU7.js +0 -867
  50. package/dist/chunk-JHOJHHU7.js.map +0 -1
  51. package/dist/chunk-JM2SKQMS.js +0 -750
  52. package/dist/chunk-JM2SKQMS.js.map +0 -1
  53. package/dist/chunk-JN2FCO5W.js +0 -7958
  54. package/dist/chunk-JN2FCO5W.js.map +0 -1
  55. package/dist/chunk-K4DBDHLK.js +0 -158
  56. package/dist/chunk-K4DBDHLK.js.map +0 -1
  57. package/dist/chunk-K6N6XJJX.js +0 -306
  58. package/dist/chunk-K6N6XJJX.js.map +0 -1
  59. package/dist/chunk-MA6HLL3S.js +0 -65
  60. package/dist/chunk-MA6HLL3S.js.map +0 -1
  61. package/dist/chunk-MAZ26DC7.js +0 -99
  62. package/dist/chunk-MAZ26DC7.js.map +0 -1
  63. package/dist/chunk-MOXWMGPC.js +0 -577
  64. package/dist/chunk-MOXWMGPC.js.map +0 -1
  65. package/dist/chunk-NJC7U437.js +0 -626
  66. package/dist/chunk-NJC7U437.js.map +0 -1
  67. package/dist/chunk-NPCTHQIO.js +0 -91
  68. package/dist/chunk-NPCTHQIO.js.map +0 -1
  69. package/dist/chunk-ONWEPEDO.js +0 -57
  70. package/dist/chunk-ONWEPEDO.js.map +0 -1
  71. package/dist/chunk-OYZAPX5G.js +0 -1526
  72. package/dist/chunk-OYZAPX5G.js.map +0 -1
  73. package/dist/chunk-PC4UYEBM.js +0 -166
  74. package/dist/chunk-PC4UYEBM.js.map +0 -1
  75. package/dist/chunk-PICTDURQ.js +0 -766
  76. package/dist/chunk-PICTDURQ.js.map +0 -1
  77. package/dist/chunk-PJQFMIOX.js +0 -1182
  78. package/dist/chunk-PJQFMIOX.js.map +0 -1
  79. package/dist/chunk-PXD6ZFNY.js +0 -1107
  80. package/dist/chunk-PXD6ZFNY.js.map +0 -1
  81. package/dist/chunk-PXE2VKMX.js +0 -140
  82. package/dist/chunk-PXE2VKMX.js.map +0 -1
  83. package/dist/chunk-PZ5AY32C.js +0 -10
  84. package/dist/chunk-PZ5AY32C.js.map +0 -1
  85. package/dist/chunk-QBRSJK47.js +0 -622
  86. package/dist/chunk-QBRSJK47.js.map +0 -1
  87. package/dist/chunk-QWMPPZ3X.js +0 -550
  88. package/dist/chunk-QWMPPZ3X.js.map +0 -1
  89. package/dist/chunk-S3UZOQ5Y.js +0 -328
  90. package/dist/chunk-S3UZOQ5Y.js.map +0 -1
  91. package/dist/chunk-S5TT5R3L.js +0 -2668
  92. package/dist/chunk-S5TT5R3L.js.map +0 -1
  93. package/dist/chunk-T4SQEITX.js +0 -95
  94. package/dist/chunk-T4SQEITX.js.map +0 -1
  95. package/dist/chunk-TT4KNT67.js +0 -124
  96. package/dist/chunk-TT4KNT67.js.map +0 -1
  97. package/dist/chunk-U5CHZ5M3.js +0 -357
  98. package/dist/chunk-U5CHZ5M3.js.map +0 -1
  99. package/dist/chunk-ULOKLHIQ.js +0 -1937
  100. package/dist/chunk-ULOKLHIQ.js.map +0 -1
  101. package/dist/chunk-VI2UW6B6.js +0 -162
  102. package/dist/chunk-VI2UW6B6.js.map +0 -1
  103. package/dist/chunk-VQMK5FMP.js +0 -247
  104. package/dist/chunk-VQMK5FMP.js.map +0 -1
  105. package/dist/chunk-VSMTAMNK.js +0 -53
  106. package/dist/chunk-VSMTAMNK.js.map +0 -1
  107. package/dist/chunk-VZSRQ272.js +0 -149
  108. package/dist/chunk-VZSRQ272.js.map +0 -1
  109. package/dist/chunk-WW2A73HW.js +0 -159
  110. package/dist/chunk-WW2A73HW.js.map +0 -1
  111. package/dist/chunk-X4UCIOTZ.js +0 -136
  112. package/dist/chunk-X4UCIOTZ.js.map +0 -1
  113. package/dist/chunk-XDIRG3TO.js +0 -1266
  114. package/dist/chunk-XDIRG3TO.js.map +0 -1
  115. package/dist/chunk-XJYR7XFV.js +0 -317
  116. package/dist/chunk-XJYR7XFV.js.map +0 -1
  117. package/dist/chunk-ZET2UAYW.js +0 -89
  118. package/dist/chunk-ZET2UAYW.js.map +0 -1
  119. package/dist/chunk-ZZUXHH3R.js +0 -99
  120. package/dist/chunk-ZZUXHH3R.js.map +0 -1
  121. package/dist/cli.d.ts +0 -1
  122. package/dist/cli.js +0 -112
  123. package/dist/cli.js.map +0 -1
  124. package/dist/contract/index.d.ts +0 -4969
  125. package/dist/contract/index.js +0 -1653
  126. package/dist/contract/index.js.map +0 -1
  127. package/dist/control.d.ts +0 -1013
  128. package/dist/control.js +0 -34
  129. package/dist/control.js.map +0 -1
  130. package/dist/fuzz.d.ts +0 -759
  131. package/dist/fuzz.js +0 -714
  132. package/dist/fuzz.js.map +0 -1
  133. package/dist/hosted/index.d.ts +0 -730
  134. package/dist/hosted/index.js +0 -14
  135. package/dist/hosted/index.js.map +0 -1
  136. package/dist/index.d.ts +0 -16780
  137. package/dist/index.js +0 -12168
  138. package/dist/index.js.map +0 -1
  139. package/dist/matrix/index.d.ts +0 -155
  140. package/dist/matrix/index.js +0 -8
  141. package/dist/matrix/index.js.map +0 -1
  142. package/dist/meta-eval/index.d.ts +0 -1030
  143. package/dist/meta-eval/index.js +0 -417
  144. package/dist/meta-eval/index.js.map +0 -1
  145. package/dist/multishot/index.d.ts +0 -579
  146. package/dist/multishot/index.js +0 -589
  147. package/dist/multishot/index.js.map +0 -1
  148. package/dist/openapi.json +0 -992
  149. package/dist/pipelines/index.d.ts +0 -567
  150. package/dist/pipelines/index.js +0 -515
  151. package/dist/pipelines/index.js.map +0 -1
  152. package/dist/reporting.d.ts +0 -1277
  153. package/dist/reporting.js +0 -48
  154. package/dist/reporting.js.map +0 -1
  155. package/dist/rl.d.ts +0 -4092
  156. package/dist/rl.js +0 -1724
  157. package/dist/rl.js.map +0 -1
  158. package/dist/run-campaign-HNFPJET4.js +0 -14
  159. package/dist/run-campaign-HNFPJET4.js.map +0 -1
  160. package/dist/storyboard/index.d.ts +0 -279
  161. package/dist/storyboard/index.js +0 -767
  162. package/dist/storyboard/index.js.map +0 -1
  163. package/dist/trace-attributes.d.ts +0 -52
  164. package/dist/trace-attributes.js +0 -62
  165. package/dist/trace-attributes.js.map +0 -1
  166. package/dist/traces.d.ts +0 -2343
  167. package/dist/traces.js +0 -249
  168. package/dist/traces.js.map +0 -1
  169. package/dist/wire/index.d.ts +0 -1252
  170. package/dist/wire/index.js +0 -81
  171. package/dist/wire/index.js.map +0 -1
@@ -1,880 +0,0 @@
1
- import {
2
- benjaminiHochberg,
3
- cohensD,
4
- confidenceInterval,
5
- pairedBootstrap,
6
- pairedMde,
7
- wilcoxonSignedRank
8
- } from "./chunk-PJQFMIOX.js";
9
- import {
10
- canonicalize,
11
- hashJson
12
- } from "./chunk-VSMTAMNK.js";
13
-
14
- // src/summary-report.ts
15
- function summaryTable(runs, opts = {}) {
16
- const split = opts.split ?? "holdout";
17
- const confidence = opts.confidence ?? 0.95;
18
- const fdr = opts.fdr ?? 0.05;
19
- const comparator = opts.comparator ?? null;
20
- const scoreField = split === "holdout" ? "holdoutScore" : "searchScore";
21
- const byCandidate = /* @__PURE__ */ new Map();
22
- for (const r of runs) {
23
- if (r.splitTag !== split) continue;
24
- const v = r.outcome[scoreField];
25
- if (typeof v !== "number" || !Number.isFinite(v)) continue;
26
- const bucket = byCandidate.get(r.candidateId) ?? { runs: [], scores: [] };
27
- bucket.runs.push(r);
28
- bucket.scores.push(v);
29
- byCandidate.set(r.candidateId, bucket);
30
- }
31
- const candidateIds = [...byCandidate.keys()].sort();
32
- const compRuns = comparator ? byCandidate.get(comparator) : void 0;
33
- const tentative = [];
34
- for (const id of candidateIds) {
35
- const bucket = byCandidate.get(id);
36
- const ci = confidenceInterval(bucket.scores, confidence);
37
- let rawP = Number.NaN;
38
- let d = Number.NaN;
39
- if (comparator && compRuns && id !== comparator) {
40
- const paired = pairScoresByKey(bucket.runs, compRuns.runs, scoreField);
41
- if (paired.before.length >= 6) {
42
- rawP = wilcoxonSignedRank(paired.before, paired.after).p;
43
- }
44
- d = cohensD(compRuns.scores, bucket.scores);
45
- }
46
- tentative.push({
47
- candidateId: id,
48
- n: bucket.scores.length,
49
- mean: ci.mean,
50
- ciLow: ci.lower,
51
- ciHigh: ci.upper,
52
- qValue: rawP,
53
- cohensD: d,
54
- rawP
55
- });
56
- }
57
- if (comparator) {
58
- const idxs = [];
59
- const ps = [];
60
- for (let i = 0; i < tentative.length; i++) {
61
- const r = tentative[i];
62
- if (r.candidateId === comparator) continue;
63
- if (!Number.isFinite(r.rawP)) continue;
64
- idxs.push(i);
65
- ps.push(r.rawP);
66
- }
67
- if (ps.length > 0) {
68
- const { qValues } = benjaminiHochberg(ps, fdr);
69
- for (let k = 0; k < idxs.length; k++) {
70
- tentative[idxs[k]].qValue = qValues[k];
71
- }
72
- }
73
- }
74
- const rows = tentative.map(({ rawP: _rawP, ...rest }) => rest);
75
- const markdown = renderSummaryTableMarkdown(rows, comparator, split);
76
- return { rows, comparator, split, markdown };
77
- }
78
- function pairScoresByKey(candidate, baseline, scoreField) {
79
- const baseIdx = /* @__PURE__ */ new Map();
80
- for (const r of baseline) {
81
- const v = r.outcome[scoreField];
82
- if (typeof v === "number" && Number.isFinite(v)) {
83
- baseIdx.set(`${r.experimentId}::${r.seed}`, v);
84
- }
85
- }
86
- const before = [];
87
- const after = [];
88
- for (const r of candidate) {
89
- const v = r.outcome[scoreField];
90
- if (typeof v !== "number" || !Number.isFinite(v)) continue;
91
- const key = `${r.experimentId}::${r.seed}`;
92
- const b = baseIdx.get(key);
93
- if (b === void 0) continue;
94
- before.push(b);
95
- after.push(v);
96
- }
97
- return { before, after };
98
- }
99
- function renderSummaryTableMarkdown(rows, comparator, split) {
100
- const lines = [];
101
- const cmpLabel = comparator ? ` (vs ${comparator})` : "";
102
- lines.push(`Summary Table \u2014 ${split} split${cmpLabel}`);
103
- lines.push("");
104
- lines.push("| Candidate | N | Mean | 95% CI | q (BH) | Cohen's d |");
105
- lines.push("|---|---:|---:|---|---:|---:|");
106
- for (const r of rows) {
107
- const ci = `[${fmt(r.ciLow)}, ${fmt(r.ciHigh)}]`;
108
- const q = Number.isFinite(r.qValue) ? r.qValue.toFixed(4) : "\u2014";
109
- const d = Number.isFinite(r.cohensD) ? r.cohensD.toFixed(3) : "\u2014";
110
- lines.push(`| ${r.candidateId} | ${r.n} | ${fmt(r.mean)} | ${ci} | ${q} | ${d} |`);
111
- }
112
- return lines.join("\n");
113
- }
114
- function paretoChart(runs, opts = {}) {
115
- const split = opts.split ?? "holdout";
116
- const scoreField = split === "holdout" ? "holdoutScore" : "searchScore";
117
- const buckets = /* @__PURE__ */ new Map();
118
- for (const r of runs) {
119
- if (r.splitTag !== split) continue;
120
- const v = r.outcome[scoreField];
121
- if (typeof v !== "number" || !Number.isFinite(v)) continue;
122
- const bucket = buckets.get(r.candidateId) ?? { cost: [], quality: [] };
123
- bucket.cost.push(r.costUsd);
124
- bucket.quality.push(v);
125
- buckets.set(r.candidateId, bucket);
126
- }
127
- const points = [];
128
- for (const [candidateId, bucket] of buckets.entries()) {
129
- points.push({
130
- candidateId,
131
- cost: avg(bucket.cost),
132
- quality: avg(bucket.quality),
133
- n: bucket.cost.length,
134
- onFrontier: false,
135
- gate: opts.gateDecisions?.[candidateId] ? gateLabel(opts.gateDecisions[candidateId]) : void 0
136
- });
137
- }
138
- for (const p of points) {
139
- p.onFrontier = !points.some((q) => q !== p && dominates(q, p));
140
- }
141
- return {
142
- kind: "pareto-cost-quality",
143
- split,
144
- axes: { x: "costUsd", y: "score" },
145
- points
146
- };
147
- }
148
- function dominates(a, b) {
149
- return a.cost <= b.cost && a.quality >= b.quality && (a.cost < b.cost || a.quality > b.quality);
150
- }
151
- function gateLabel(d) {
152
- if (d.promote) return "promote";
153
- if (d.rejectionCode === "few_runs") return "reject_few_runs";
154
- if (d.rejectionCode === "negative_delta") return "reject_negative_delta";
155
- if (d.rejectionCode === "overfit_gap") return "reject_overfit_gap";
156
- return null;
157
- }
158
- function gainHistogram(runs, candidateId, comparator, opts = {}) {
159
- const split = opts.split ?? "holdout";
160
- const scoreField = split === "holdout" ? "holdoutScore" : "searchScore";
161
- const binCount = opts.bins ?? 11;
162
- if (binCount < 1) throw new Error("gainHistogram: bins must be \u2265 1");
163
- const candidate = runs.filter((r) => r.candidateId === candidateId && r.splitTag === split);
164
- const baseline = runs.filter((r) => r.candidateId === comparator && r.splitTag === split);
165
- const { before, after } = pairScoresByKey(candidate, baseline, scoreField);
166
- const n = before.length;
167
- if (n === 0) {
168
- return {
169
- kind: "gain-distribution",
170
- candidateId,
171
- comparator,
172
- split,
173
- n: 0,
174
- bins: [],
175
- median: 0,
176
- ci: { low: 0, high: 0 }
177
- };
178
- }
179
- const deltas = before.map((b, i) => after[i] - b);
180
- const sortedDeltas = [...deltas].sort((a, b) => a - b);
181
- const median = medianOfSorted(sortedDeltas);
182
- const min = sortedDeltas[0];
183
- const max = sortedDeltas[sortedDeltas.length - 1];
184
- const bound = Math.max(Math.abs(min), Math.abs(max), 1e-6);
185
- const lo = -bound;
186
- const hi = bound;
187
- const width = (hi - lo) / binCount;
188
- const bins = [];
189
- for (let i = 0; i < binCount; i++) {
190
- bins.push({ lo: lo + i * width, hi: lo + (i + 1) * width, count: 0 });
191
- }
192
- for (const d of deltas) {
193
- let idx = Math.floor((d - lo) / width);
194
- if (idx < 0) idx = 0;
195
- if (idx >= binCount) idx = binCount - 1;
196
- bins[idx].count += 1;
197
- }
198
- const ci = pairedBootstrap(before, after, {
199
- confidence: opts.confidence ?? 0.95,
200
- resamples: opts.resamples ?? 2e3,
201
- statistic: "median",
202
- seed: opts.seed
203
- });
204
- return {
205
- kind: "gain-distribution",
206
- candidateId,
207
- comparator,
208
- split,
209
- n,
210
- bins,
211
- median,
212
- ci: { low: ci.low, high: ci.high }
213
- };
214
- }
215
- var RESEARCH_REPORT_HARD_PAIR_FLOOR = 6;
216
- function pairedPosterior(runs, candidateId, comparator, opts) {
217
- const scoreField = opts.split === "holdout" ? "holdoutScore" : "searchScore";
218
- const candidate = runs.filter((r) => r.candidateId === candidateId && r.splitTag === opts.split);
219
- const baseline = runs.filter((r) => r.candidateId === comparator && r.splitTag === opts.split);
220
- const { before, after } = pairScoresByKey(candidate, baseline, scoreField);
221
- const n = before.length;
222
- if (n === 0) return null;
223
- const deltas = before.map((b, i) => after[i] - b);
224
- const meanDelta = deltas.reduce((s, x) => s + x, 0) / n;
225
- const sortedDeltas = [...deltas].sort((a, b) => a - b);
226
- const medianDelta = medianOfSorted(sortedDeltas);
227
- const sdDelta = stdev(deltas, meanDelta);
228
- const ci = pairedBootstrap(before, after, {
229
- confidence: opts.confidence,
230
- resamples: 2e3,
231
- statistic: "median",
232
- seed: opts.seed
233
- });
234
- const meanSamples = bootstrapMeanSamples(deltas, 2e3, opts.seed);
235
- const prGreaterThanZero = meanSamples.length === 0 ? 0 : meanSamples.filter((s) => s > 0).length / meanSamples.length;
236
- const prInRope = opts.rope === null || meanSamples.length === 0 ? null : meanSamples.filter((s) => s >= opts.rope.low && s <= opts.rope.high).length / meanSamples.length;
237
- const dStandardised = pairedMde({ nPaired: n, alpha: opts.mdeAlpha, power: opts.mdePower });
238
- const mde = sdDelta === 0 ? 0 : dStandardised * sdDelta;
239
- return {
240
- n,
241
- meanDelta,
242
- medianDelta,
243
- sdDelta,
244
- ci: { low: ci.low, high: ci.high },
245
- prGreaterThanZero,
246
- prInRope,
247
- mde
248
- };
249
- }
250
- function bootstrapMeanSamples(deltas, resamples, seed) {
251
- const n = deltas.length;
252
- if (n === 0) return [];
253
- if (n === 1) return new Array(resamples).fill(deltas[0]);
254
- const rng = seedRng(seed);
255
- const samples = new Array(resamples);
256
- for (let b = 0; b < resamples; b++) {
257
- let sum = 0;
258
- for (let k = 0; k < n; k++) sum += deltas[Math.floor(rng() * n)];
259
- samples[b] = sum / n;
260
- }
261
- return samples;
262
- }
263
- function seedRng(seed) {
264
- if (seed === void 0) return Math.random;
265
- let s = seed >>> 0;
266
- return () => {
267
- s = s + 1831565813 >>> 0;
268
- let t = s;
269
- t = Math.imul(t ^ t >>> 15, t | 1);
270
- t ^= t + Math.imul(t ^ t >>> 7, t | 61);
271
- return ((t ^ t >>> 14) >>> 0) / 4294967296;
272
- };
273
- }
274
- function stdev(xs, mean) {
275
- if (xs.length < 2) return 0;
276
- let sse = 0;
277
- for (const x of xs) sse += (x - mean) ** 2;
278
- return Math.sqrt(sse / (xs.length - 1));
279
- }
280
- async function researchReport(runs, opts = {}) {
281
- const split = opts.split ?? "holdout";
282
- const comparator = opts.comparator ?? null;
283
- const confidence = opts.confidence ?? 0.95;
284
- const fdr = opts.fdr ?? 0.05;
285
- const minPairs = Math.max(opts.minPairs ?? 20, RESEARCH_REPORT_HARD_PAIR_FLOOR);
286
- const rope = opts.rope ?? null;
287
- const mdePower = opts.mdePower ?? 0.8;
288
- const mdeAlpha = opts.mdeAlpha ?? fdr;
289
- const title = opts.title ?? "Agent Evaluation Research Report";
290
- const generatedAt = opts.generatedAt ?? (/* @__PURE__ */ new Date()).toISOString();
291
- const preregistrationHash = opts.preregistrationHash ?? null;
292
- if (rope && !(Number.isFinite(rope.low) && Number.isFinite(rope.high) && rope.low <= rope.high)) {
293
- throw new Error(
294
- `researchReport: rope must satisfy low \u2264 high with finite bounds, got ${JSON.stringify(rope)}`
295
- );
296
- }
297
- const summary = summaryTable(runs, {
298
- comparator: comparator ?? void 0,
299
- split,
300
- confidence,
301
- fdr
302
- });
303
- const pareto = paretoChart(runs, { split, gateDecisions: opts.gateDecisions });
304
- const candidateIds = opts.candidateIds ?? summary.rows.map((r) => r.candidateId).filter((id) => id !== comparator);
305
- const gains = comparator ? candidateIds.map(
306
- (id) => gainHistogram(runs, id, comparator, {
307
- split,
308
- confidence,
309
- seed: opts.seed
310
- })
311
- ) : [];
312
- const gainByCandidate = new Map(gains.map((g) => [g.candidateId, g]));
313
- const paretoByCandidate = new Map(pareto.points.map((p) => [p.candidateId, p]));
314
- const posteriorByCandidate = /* @__PURE__ */ new Map();
315
- if (comparator) {
316
- for (const id of candidateIds) {
317
- posteriorByCandidate.set(
318
- id,
319
- pairedPosterior(runs, id, comparator, {
320
- split,
321
- confidence,
322
- seed: opts.seed,
323
- rope,
324
- mdePower,
325
- mdeAlpha
326
- })
327
- );
328
- }
329
- }
330
- const candidates = summary.rows.map((row) => {
331
- const gain = gainByCandidate.get(row.candidateId);
332
- const point = paretoByCandidate.get(row.candidateId);
333
- const posterior = posteriorByCandidate.get(row.candidateId) ?? null;
334
- const classified = classifyCandidate(row, {
335
- comparator,
336
- posterior,
337
- point,
338
- fdr,
339
- minPairs,
340
- rope
341
- });
342
- return {
343
- candidateId: row.candidateId,
344
- n: row.n,
345
- mean: row.mean,
346
- ciLow: row.ciLow,
347
- ciHigh: row.ciHigh,
348
- qValue: row.qValue,
349
- cohensD: row.cohensD,
350
- meanDeltaVsComparator: posterior ? posterior.meanDelta : null,
351
- pairedN: posterior?.n ?? gain?.n ?? 0,
352
- medianGain: posterior ? posterior.medianDelta : gain ? gain.median : null,
353
- meanGain: posterior ? posterior.meanDelta : null,
354
- gainCi: posterior ? posterior.ci : gain ? gain.ci : null,
355
- prGreaterThanZero: posterior ? posterior.prGreaterThanZero : null,
356
- prInRope: posterior ? posterior.prInRope : null,
357
- mde: posterior ? posterior.mde : null,
358
- onParetoFrontier: point?.onFrontier ?? false,
359
- gate: point?.gate,
360
- decision: classified.decision,
361
- decisionReason: classified.reason
362
- };
363
- }).sort((a, b) => {
364
- const decisionRank = decisionWeight(b.decision) - decisionWeight(a.decision);
365
- if (decisionRank !== 0) return decisionRank;
366
- return b.mean - a.mean;
367
- });
368
- const recommendation = buildRecommendation(candidates, {
369
- comparator,
370
- failureClusters: opts.failureClusters,
371
- rope,
372
- minPairs,
373
- preregistrationHash
374
- });
375
- const executiveSummary = buildExecutiveSummary(candidates, recommendation, {
376
- comparator,
377
- split,
378
- failureClusters: opts.failureClusters,
379
- preregistrationHash
380
- });
381
- const methodology = buildMethodology({
382
- split,
383
- comparator,
384
- fdr,
385
- minPairs,
386
- rope,
387
- confidence,
388
- mdePower,
389
- mdeAlpha
390
- });
391
- const runFingerprint = await hashJson(
392
- canonicalize({
393
- triples: runs.filter((r) => r.splitTag === split).map((r) => ({ runId: r.runId, candidateId: r.candidateId, splitTag: r.splitTag })).sort((a, b) => a.runId.localeCompare(b.runId)),
394
- comparator,
395
- split
396
- })
397
- );
398
- const markdown = renderResearchMarkdown({
399
- title,
400
- generatedAt,
401
- split,
402
- comparator,
403
- rope,
404
- runFingerprint,
405
- preregistrationHash,
406
- executiveSummary,
407
- recommendation,
408
- candidates,
409
- summary,
410
- pareto,
411
- gains,
412
- methodology,
413
- failureClusters: opts.failureClusters
414
- });
415
- const html = renderResearchHtml(markdown, title);
416
- return {
417
- kind: "agent-eval-research-report",
418
- title,
419
- generatedAt,
420
- split,
421
- comparator,
422
- runFingerprint,
423
- preregistrationHash,
424
- rope,
425
- executiveSummary,
426
- recommendation,
427
- candidates,
428
- summary,
429
- charts: { pareto, gains },
430
- methodology,
431
- failureClusters: opts.failureClusters,
432
- markdown,
433
- html
434
- };
435
- }
436
- function buildMethodology(ctx) {
437
- const assumptions = [
438
- "Pairs are matched by (experimentId, seed); the candidate and comparator see the same scenarios in the same order.",
439
- "Paired deltas are exchangeable conditional on the matched scenario \u2014 no mid-run distribution shift.",
440
- `Decisions are pre-specified at fdr=${ctx.fdr}, minPairs=${ctx.minPairs}, confidence=${ctx.confidence}; deviating from these post-hoc invalidates the false-discovery control.`
441
- ];
442
- if (ctx.rope) {
443
- assumptions.push(
444
- `The Region of Practical Equivalence ${formatRope(ctx.rope)} is supplied by the domain owner; equivalent verdicts are only meaningful if that range is treated as the standing definition of "no material difference."`
445
- );
446
- }
447
- if (ctx.comparator === null) {
448
- assumptions.push("No comparator was configured; this run is descriptive, not causal.");
449
- }
450
- const methods = [
451
- "Marginal scores summarised with BH-FDR-adjusted Wilcoxon signed-rank q-values and Cohen's d via summaryTable.",
452
- "Paired evidence summarised with bootstrap CI on the median delta and Bayesian-bootstrap-style Pr(\u0394>0) and Pr(\u0394\u2208ROPE) on the mean delta.",
453
- `Minimum detectable effect reported per candidate at \u03B1=${ctx.mdeAlpha} (two-sided), power=${ctx.mdePower}, standardised by the observed paired-delta SD.`,
454
- "Pareto frontier flagged as a separate axis (cost vs quality); a candidate can be on-frontier without winning the paired test.",
455
- "Held-out gate decisions, when supplied, override the statistical verdict in the reject direction."
456
- ];
457
- const alternatives = [
458
- "Paired t-test rejected: not robust to the heavy-tailed score distributions common in agent benchmarks.",
459
- "Unpaired Mann\u2013Whitney rejected: matched scenarios make pairing free; unpaired throws away that variance reduction.",
460
- "Sequential / always-valid inference (e-values, mSPRT) is the right tool for iterative sweeps and is out of scope for this single-look report \u2014 preregister and run once, or wrap this report in an alpha-spending schedule.",
461
- "Hierarchical Bayesian shrinkage across many candidates is future work; the current ranking uses raw paired statistics."
462
- ];
463
- const whenNotToApply = [
464
- `Paired N below ${RESEARCH_REPORT_HARD_PAIR_FLOOR} on any candidate \u2014 the bootstrap CI is degenerate.`,
465
- "Comparator chosen post-hoc by inspecting the same data; q-values are no longer false-discovery-controlled.",
466
- "Scenarios not drawn under a stable preregistered protocol; the report can describe the data but cannot anchor a launch decision.",
467
- "Score distributions with mid-run shift (judge model swap, rubric change, infra outage) \u2014 pair exchangeability is violated."
468
- ];
469
- const citations = [
470
- "Benjamini, Y. & Hochberg, Y. (1995). Controlling the false discovery rate: a practical and powerful approach to multiple testing. JRSS B, 57(1), 289\u2013300.",
471
- "Wilcoxon, F. (1945). Individual comparisons by ranking methods. Biometrics Bulletin, 1(6), 80\u201383.",
472
- "Efron, B. (1979). Bootstrap methods: another look at the jackknife. Annals of Statistics, 7(1), 1\u201326.",
473
- "Rubin, D. B. (1981). The Bayesian bootstrap. Annals of Statistics, 9(1), 130\u2013134.",
474
- "Kruschke, J. K. (2018). Rejecting or accepting parameter values in Bayesian estimation. Advances in Methods and Practices in Psychological Science, 1(2), 270\u2013280. (ROPE.)"
475
- ];
476
- return { assumptions, methods, alternatives, whenNotToApply, citations };
477
- }
478
- function formatRope(rope) {
479
- return `[${fmt(rope.low)}, ${fmt(rope.high)}]`;
480
- }
481
- function classifyCandidate(row, ctx) {
482
- if (ctx.comparator && row.candidateId === ctx.comparator) {
483
- return { decision: "hold", reason: "Comparator baseline." };
484
- }
485
- if (!ctx.comparator) {
486
- return {
487
- decision: ctx.point?.onFrontier ? "hold" : "needs_more_data",
488
- reason: "No comparator configured; report ranks candidates but cannot anchor a promotion call."
489
- };
490
- }
491
- if (ctx.point?.gate && ctx.point.gate !== "promote") {
492
- return { decision: "reject", reason: `Held-out gate returned ${ctx.point.gate}.` };
493
- }
494
- if (!ctx.posterior || ctx.posterior.n < RESEARCH_REPORT_HARD_PAIR_FLOOR) {
495
- return {
496
- decision: "needs_more_data",
497
- reason: `Only ${ctx.posterior?.n ?? 0} paired observations; below hard floor of ${RESEARCH_REPORT_HARD_PAIR_FLOOR} for any paired inference.`
498
- };
499
- }
500
- const ci = ctx.posterior.ci;
501
- if (ctx.rope && ci.low >= ctx.rope.low && ci.high <= ctx.rope.high) {
502
- return {
503
- decision: "equivalent",
504
- reason: `Paired-delta CI [${fmt(ci.low)}, ${fmt(ci.high)}] is fully inside ROPE ${formatRope(ctx.rope)}; candidate is practically equivalent to comparator.`
505
- };
506
- }
507
- const significant = Number.isFinite(row.qValue) && row.qValue <= ctx.fdr;
508
- const gainPositive = ci.low > 0;
509
- const gainNegative = ci.high < 0;
510
- if (gainNegative) {
511
- return {
512
- decision: "reject",
513
- reason: `Paired-delta CI [${fmt(ci.low)}, ${fmt(ci.high)}] lies entirely below zero.`
514
- };
515
- }
516
- if (ctx.posterior.n < ctx.minPairs) {
517
- return {
518
- decision: "needs_more_data",
519
- reason: `Only ${ctx.posterior.n} paired observations; minimum detectable effect at this N is ${fmt(ctx.posterior.mde)} score units (need \u2265 ${ctx.minPairs} pairs to issue a directional verdict).`
520
- };
521
- }
522
- if (significant && gainPositive) {
523
- return {
524
- decision: "promote",
525
- reason: `BH-adjusted q=${fmt(row.qValue)} \u2264 ${ctx.fdr} and paired-delta CI [${fmt(ci.low)}, ${fmt(ci.high)}] excludes zero; Pr(\u0394>0)=${fmt(ctx.posterior.prGreaterThanZero)}.`
526
- };
527
- }
528
- return {
529
- decision: "hold",
530
- reason: `Pr(\u0394>0)=${fmt(ctx.posterior.prGreaterThanZero)} but CI [${fmt(ci.low)}, ${fmt(ci.high)}] crosses zero; effect not decisive at fdr=${ctx.fdr}.`
531
- };
532
- }
533
- function buildRecommendation(candidates, ctx) {
534
- const nonComparator = candidates.filter((c) => c.candidateId !== ctx.comparator);
535
- const bestPromote = nonComparator.find((c) => c.decision === "promote");
536
- const bestEquivalent = nonComparator.find((c) => c.decision === "equivalent");
537
- const chosen = bestPromote ?? bestEquivalent ?? nonComparator[0] ?? null;
538
- const decision = bestPromote ? "promote" : nonComparator.some((c) => c.decision === "needs_more_data") ? "needs_more_data" : bestEquivalent ? "equivalent" : nonComparator.some((c) => c.decision === "hold") ? "hold" : "reject";
539
- const rationale = [];
540
- const risks = [];
541
- const nextActions = [];
542
- if (chosen) {
543
- rationale.push(`${chosen.candidateId}: ${chosen.decisionReason}`);
544
- if (chosen.gainCi) {
545
- const probSummary = chosen.prGreaterThanZero !== null ? `, Pr(\u0394>0)=${fmt(chosen.prGreaterThanZero)}` : "";
546
- rationale.push(
547
- `Median paired gain CI: [${fmt(chosen.gainCi.low)}, ${fmt(chosen.gainCi.high)}]${probSummary}.`
548
- );
549
- }
550
- if (chosen.mde !== null && Number.isFinite(chosen.mde)) {
551
- rationale.push(`MDE at current paired N=${chosen.pairedN}: ${fmt(chosen.mde)} score units.`);
552
- }
553
- }
554
- if (!ctx.comparator) {
555
- risks.push("No comparator was configured; verdict is descriptive, not causal.");
556
- nextActions.push("Re-run with a stable comparator candidate for paired inference.");
557
- }
558
- if (!ctx.preregistrationHash) {
559
- risks.push(
560
- "No preregistration hash supplied; readers cannot verify the analysis was specified before data inspection."
561
- );
562
- nextActions.push(
563
- "Sign a HypothesisManifest before the next sweep and pass `preregistrationHash` so the report cites it."
564
- );
565
- }
566
- if (ctx.rope === null && nonComparator.length > 0) {
567
- risks.push(
568
- 'No ROPE configured; the report cannot distinguish "equivalent" from "inconclusive".'
569
- );
570
- nextActions.push(
571
- "Define a domain-specific Region of Practical Equivalence and pass it to lock in the equivalence threshold."
572
- );
573
- }
574
- const inconclusive = nonComparator.filter((c) => c.decision === "needs_more_data");
575
- if (inconclusive.length > 0) {
576
- const worst = inconclusive.reduce((a, b) => b.pairedN < a.pairedN ? b : a);
577
- risks.push(
578
- `${inconclusive.length} candidate(s) below soft floor (${ctx.minPairs} pairs); thinnest is ${worst.candidateId} with ${worst.pairedN}.`
579
- );
580
- nextActions.push(
581
- `Collect at least ${ctx.minPairs - worst.pairedN} more matched holdout runs for ${worst.candidateId}.`
582
- );
583
- }
584
- const rejected = nonComparator.filter((c) => c.decision === "reject");
585
- if (rejected.length > 0) {
586
- risks.push(
587
- `${rejected.length} candidate(s) failed the paired test or held-out gate; do not ship those variants.`
588
- );
589
- }
590
- if (ctx.failureClusters && ctx.failureClusters.clusters.length > 0) {
591
- const top = ctx.failureClusters.clusters[0];
592
- risks.push(`Top failure cluster: ${top.failureClass} across ${top.runCount} run(s).`);
593
- nextActions.push("Prioritize the largest failure cluster before broad rollout.");
594
- }
595
- if (decision === "promote") {
596
- nextActions.push("Ship behind the existing promotion gate and monitor canaries.");
597
- } else if (decision === "hold") {
598
- nextActions.push("Keep current production candidate while expanding holdout evidence.");
599
- } else if (decision === "equivalent") {
600
- nextActions.push(
601
- "Either keep the comparator (no quality regression) or promote on cost/latency grounds \u2014 equivalence does not justify either; the choice is a product decision, not a stats one."
602
- );
603
- } else if (decision === "reject") {
604
- nextActions.push(
605
- "Do not promote this sweep; inspect failures and generate a revised candidate."
606
- );
607
- }
608
- return {
609
- decision,
610
- candidateId: chosen?.candidateId ?? null,
611
- rationale,
612
- risks,
613
- nextActions
614
- };
615
- }
616
- function buildExecutiveSummary(candidates, recommendation, ctx) {
617
- const lines = [];
618
- const nonComparator = candidates.filter((c) => c.candidateId !== ctx.comparator);
619
- lines.push(
620
- `Evaluated ${nonComparator.length} candidate(s) on the ${ctx.split} split${ctx.comparator ? ` against ${ctx.comparator}` : ""}.`
621
- );
622
- lines.push(
623
- `Recommendation: ${recommendation.decision}${recommendation.candidateId ? ` ${recommendation.candidateId}` : ""}.`
624
- );
625
- const promoted = nonComparator.filter((c) => c.decision === "promote").length;
626
- const held = nonComparator.filter((c) => c.decision === "hold").length;
627
- const equivalent = nonComparator.filter((c) => c.decision === "equivalent").length;
628
- const rejected = nonComparator.filter((c) => c.decision === "reject").length;
629
- const more = nonComparator.filter((c) => c.decision === "needs_more_data").length;
630
- lines.push(
631
- `Decision mix: ${promoted} promote, ${equivalent} equivalent, ${held} hold, ${rejected} reject, ${more} need more data.`
632
- );
633
- const frontier = nonComparator.filter((c) => c.onParetoFrontier).map((c) => c.candidateId);
634
- if (frontier.length > 0) lines.push(`Pareto-frontier candidates: ${frontier.join(", ")}.`);
635
- if (ctx.failureClusters) {
636
- lines.push(
637
- `Failure clustering found ${ctx.failureClusters.totalFailures}/${ctx.failureClusters.totalRuns} failed runs across ${ctx.failureClusters.clusters.length} reportable cluster(s).`
638
- );
639
- }
640
- lines.push(
641
- ctx.preregistrationHash ? `Preregistered analysis: ${ctx.preregistrationHash.slice(0, 12)}\u2026` : "Analysis is post-hoc \u2014 no preregistration hash supplied."
642
- );
643
- return lines;
644
- }
645
- function renderResearchMarkdown(report) {
646
- const lines = [];
647
- lines.push(`# ${report.title}`);
648
- lines.push("");
649
- lines.push(`**Generated:** ${report.generatedAt}`);
650
- lines.push(`**Primary split:** ${report.split}`);
651
- lines.push(`**Comparator:** ${report.comparator ?? "not configured"}`);
652
- lines.push(`**ROPE:** ${report.rope ? formatRope(report.rope) : "not configured"}`);
653
- lines.push(`**Run fingerprint:** \`${report.runFingerprint}\``);
654
- lines.push(
655
- `**Preregistration:** ${report.preregistrationHash ? `\`${report.preregistrationHash}\`` : "none"}`
656
- );
657
- lines.push("");
658
- lines.push("## Executive Summary");
659
- lines.push("");
660
- for (const item of report.executiveSummary) lines.push(`- ${item}`);
661
- lines.push("");
662
- lines.push("## Recommendation");
663
- lines.push("");
664
- lines.push(`**Decision:** ${report.recommendation.decision}`);
665
- lines.push(`**Candidate:** ${report.recommendation.candidateId ?? "N/A"}`);
666
- lines.push("");
667
- lines.push("### Rationale");
668
- lines.push("");
669
- for (const item of report.recommendation.rationale) lines.push(`- ${item}`);
670
- lines.push("");
671
- lines.push("### Risks");
672
- lines.push("");
673
- for (const item of report.recommendation.risks.length ? report.recommendation.risks : ["No material report-level risks detected."]) {
674
- lines.push(`- ${item}`);
675
- }
676
- lines.push("");
677
- lines.push("### Next Actions");
678
- lines.push("");
679
- for (const item of report.recommendation.nextActions) lines.push(`- ${item}`);
680
- lines.push("");
681
- lines.push("## Candidate Decision Table");
682
- lines.push("");
683
- lines.push(
684
- "| Candidate | Decision | Mean | \u0394\u0304 | Pr(\u0394>0) | q | d | Paired N | Median Gain CI | MDE | Pareto | Gate |"
685
- );
686
- lines.push("|---|---|---:|---:|---:|---:|---:|---:|---|---:|---|---|");
687
- for (const c of report.candidates) {
688
- const delta = c.meanDeltaVsComparator === null ? "-" : signed(c.meanDeltaVsComparator);
689
- const prGt = c.prGreaterThanZero === null ? "-" : c.prGreaterThanZero.toFixed(3);
690
- const q = Number.isFinite(c.qValue) ? c.qValue.toFixed(4) : "-";
691
- const d = Number.isFinite(c.cohensD) ? c.cohensD.toFixed(3) : "-";
692
- const gain = c.gainCi ? `[${fmt(c.gainCi.low)}, ${fmt(c.gainCi.high)}]` : "-";
693
- const mde = c.mde === null || !Number.isFinite(c.mde) ? "-" : fmt(c.mde);
694
- lines.push(
695
- `| ${c.candidateId} | ${c.decision} | ${fmt(c.mean)} | ${delta} | ${prGt} | ${q} | ${d} | ${c.pairedN} | ${gain} | ${mde} | ${c.onParetoFrontier ? "yes" : "no"} | ${c.gate ?? "-"} |`
696
- );
697
- }
698
- lines.push("");
699
- lines.push("## Statistical Summary");
700
- lines.push("");
701
- lines.push(report.summary.markdown);
702
- lines.push("");
703
- lines.push("## Methodology");
704
- lines.push("");
705
- lines.push("### Assumptions");
706
- lines.push("");
707
- for (const item of report.methodology.assumptions) lines.push(`- ${item}`);
708
- lines.push("");
709
- lines.push("### Methods");
710
- lines.push("");
711
- for (const item of report.methodology.methods) lines.push(`- ${item}`);
712
- lines.push("");
713
- lines.push("### Alternatives Considered");
714
- lines.push("");
715
- for (const item of report.methodology.alternatives) lines.push(`- ${item}`);
716
- lines.push("");
717
- lines.push("### When NOT To Apply");
718
- lines.push("");
719
- for (const item of report.methodology.whenNotToApply) lines.push(`- ${item}`);
720
- lines.push("");
721
- lines.push("### Citations");
722
- lines.push("");
723
- for (const item of report.methodology.citations) lines.push(`- ${item}`);
724
- lines.push("");
725
- lines.push("## Chart Specs");
726
- lines.push("");
727
- lines.push(
728
- "The report carries JSON chart specs for Pareto cost/quality and paired gain histograms."
729
- );
730
- lines.push("");
731
- lines.push("```json");
732
- lines.push(JSON.stringify({ pareto: report.pareto, gains: report.gains }, null, 2));
733
- lines.push("```");
734
- if (report.failureClusters) {
735
- lines.push("");
736
- lines.push("## Failure Clusters");
737
- lines.push("");
738
- lines.push("| Failure Class | Runs | Scenarios | Tool | Example |");
739
- lines.push("|---|---:|---:|---|---|");
740
- for (const c of report.failureClusters.clusters.slice(0, 10)) {
741
- lines.push(
742
- `| ${c.failureClass} | ${c.runCount} | ${c.scenarioIds.length} | ${c.toolName ?? "-"} | ${escapePipes(c.exampleError ?? c.exampleRunId)} |`
743
- );
744
- }
745
- }
746
- return lines.join("\n");
747
- }
748
- function renderResearchHtml(markdown, title) {
749
- const body = markdownToHtml(markdown);
750
- return [
751
- "<!doctype html>",
752
- '<html lang="en">',
753
- "<head>",
754
- '<meta charset="utf-8">',
755
- '<meta name="viewport" content="width=device-width, initial-scale=1">',
756
- `<title>${escapeHtml(title)}</title>`,
757
- "<style>",
758
- 'body{font-family:Inter,ui-sans-serif,system-ui,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif;margin:0;color:#172026;background:#f7f8f8;}',
759
- "main{max-width:1080px;margin:0 auto;padding:40px 24px 64px;background:#fff;min-height:100vh;}",
760
- "h1{font-size:34px;line-height:1.15;margin:0 0 20px;}h2{margin-top:34px;border-top:1px solid #d9dfdf;padding-top:22px;}h3{margin-top:22px;}",
761
- "p,li{line-height:1.55;}table{border-collapse:collapse;width:100%;margin:16px 0;font-size:14px;}th,td{border:1px solid #d9dfdf;padding:8px;text-align:left;}th{background:#eef2f2;}",
762
- "code,pre{font-family:ui-monospace,SFMono-Regular,Menlo,monospace;}pre{overflow:auto;background:#111827;color:#f9fafb;padding:16px;border-radius:6px;}",
763
- "</style>",
764
- "</head>",
765
- "<body><main>",
766
- body,
767
- "</main></body></html>"
768
- ].join("\n");
769
- }
770
- function markdownToHtml(markdown) {
771
- const lines = markdown.split("\n");
772
- const html = [];
773
- let inList = false;
774
- let inCode = false;
775
- let code = [];
776
- let table = [];
777
- const flushList = () => {
778
- if (inList) {
779
- html.push("</ul>");
780
- inList = false;
781
- }
782
- };
783
- const flushTable = () => {
784
- if (table.length === 0) return;
785
- html.push(renderMarkdownTable(table));
786
- table = [];
787
- };
788
- for (const line of lines) {
789
- if (line.startsWith("```")) {
790
- if (inCode) {
791
- html.push(`<pre><code>${escapeHtml(code.join("\n"))}</code></pre>`);
792
- code = [];
793
- inCode = false;
794
- } else {
795
- flushList();
796
- flushTable();
797
- inCode = true;
798
- }
799
- continue;
800
- }
801
- if (inCode) {
802
- code.push(line);
803
- continue;
804
- }
805
- if (line.startsWith("|")) {
806
- flushList();
807
- table.push(line);
808
- continue;
809
- }
810
- flushTable();
811
- if (line.startsWith("- ")) {
812
- if (!inList) {
813
- html.push("<ul>");
814
- inList = true;
815
- }
816
- html.push(`<li>${inlineMarkdown(line.slice(2))}</li>`);
817
- continue;
818
- }
819
- flushList();
820
- if (line.startsWith("# ")) html.push(`<h1>${inlineMarkdown(line.slice(2))}</h1>`);
821
- else if (line.startsWith("## ")) html.push(`<h2>${inlineMarkdown(line.slice(3))}</h2>`);
822
- else if (line.startsWith("### ")) html.push(`<h3>${inlineMarkdown(line.slice(4))}</h3>`);
823
- else if (line.trim() === "") html.push("");
824
- else html.push(`<p>${inlineMarkdown(line)}</p>`);
825
- }
826
- flushList();
827
- flushTable();
828
- return html.join("\n");
829
- }
830
- function renderMarkdownTable(lines) {
831
- const rows = lines.filter((line) => !/^\|[-:\s|]+\|$/.test(line)).map(
832
- (line) => line.slice(1, -1).split("|").map((cell) => inlineMarkdown(cell.trim()))
833
- );
834
- if (rows.length === 0) return "";
835
- const [head, ...body] = rows;
836
- const th = head.map((cell) => `<th>${cell}</th>`).join("");
837
- const trs = body.map((row) => `<tr>${row.map((cell) => `<td>${cell}</td>`).join("")}</tr>`).join("\n");
838
- return `<table><thead><tr>${th}</tr></thead><tbody>${trs}</tbody></table>`;
839
- }
840
- function inlineMarkdown(s) {
841
- return escapeHtml(s).replace(/\*\*([^*]+)\*\*/g, "<strong>$1</strong>");
842
- }
843
- function escapeHtml(s) {
844
- return s.replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;").replace(/"/g, "&quot;");
845
- }
846
- function escapePipes(s) {
847
- return s.replace(/\|/g, "\\|");
848
- }
849
- function decisionWeight(decision) {
850
- if (decision === "promote") return 5;
851
- if (decision === "equivalent") return 4;
852
- if (decision === "hold") return 3;
853
- if (decision === "needs_more_data") return 2;
854
- return 1;
855
- }
856
- function signed(x) {
857
- return `${x >= 0 ? "+" : ""}${fmt(x)}`;
858
- }
859
- function avg(xs) {
860
- if (xs.length === 0) return Number.NaN;
861
- return xs.reduce((s, x) => s + x, 0) / xs.length;
862
- }
863
- function medianOfSorted(sorted) {
864
- if (sorted.length === 0) return 0;
865
- const mid = Math.floor(sorted.length / 2);
866
- return sorted.length % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
867
- }
868
- function fmt(x) {
869
- if (!Number.isFinite(x)) return String(x);
870
- return x.toFixed(4);
871
- }
872
-
873
- export {
874
- summaryTable,
875
- paretoChart,
876
- gainHistogram,
877
- RESEARCH_REPORT_HARD_PAIR_FLOOR,
878
- researchReport
879
- };
880
- //# sourceMappingURL=chunk-DPZAEKA6.js.map