@tangle-network/agent-eval 0.120.0 → 0.120.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/package.json +1 -1
  3. package/dist/analyst/index.d.ts +0 -3111
  4. package/dist/analyst/index.js +0 -403
  5. package/dist/analyst/index.js.map +0 -1
  6. package/dist/authenticity/index.d.ts +0 -161
  7. package/dist/authenticity/index.js +0 -215
  8. package/dist/authenticity/index.js.map +0 -1
  9. package/dist/belief-state/index.d.ts +0 -1301
  10. package/dist/belief-state/index.js +0 -2152
  11. package/dist/belief-state/index.js.map +0 -1
  12. package/dist/benchmarks/index.d.ts +0 -974
  13. package/dist/benchmarks/index.js +0 -60
  14. package/dist/benchmarks/index.js.map +0 -1
  15. package/dist/builder-eval/index.d.ts +0 -695
  16. package/dist/builder-eval/index.js +0 -366
  17. package/dist/builder-eval/index.js.map +0 -1
  18. package/dist/campaign/index.d.ts +0 -7454
  19. package/dist/campaign/index.js +0 -272
  20. package/dist/campaign/index.js.map +0 -1
  21. package/dist/chunk-32BZXMSO.js +0 -3878
  22. package/dist/chunk-32BZXMSO.js.map +0 -1
  23. package/dist/chunk-3A246TSA.js +0 -998
  24. package/dist/chunk-3A246TSA.js.map +0 -1
  25. package/dist/chunk-3RF76KTD.js +0 -84
  26. package/dist/chunk-3RF76KTD.js.map +0 -1
  27. package/dist/chunk-3YYRZDON.js +0 -45
  28. package/dist/chunk-3YYRZDON.js.map +0 -1
  29. package/dist/chunk-4I2E3LLO.js +0 -1030
  30. package/dist/chunk-4I2E3LLO.js.map +0 -1
  31. package/dist/chunk-ARU2PZFM.js +0 -312
  32. package/dist/chunk-ARU2PZFM.js.map +0 -1
  33. package/dist/chunk-BOD4O7OF.js +0 -40
  34. package/dist/chunk-BOD4O7OF.js.map +0 -1
  35. package/dist/chunk-DPZAEKA6.js +0 -880
  36. package/dist/chunk-DPZAEKA6.js.map +0 -1
  37. package/dist/chunk-DTJ6QUQB.js +0 -131
  38. package/dist/chunk-DTJ6QUQB.js.map +0 -1
  39. package/dist/chunk-GGE4NNQT.js +0 -65
  40. package/dist/chunk-GGE4NNQT.js.map +0 -1
  41. package/dist/chunk-H5UD2323.js +0 -286
  42. package/dist/chunk-H5UD2323.js.map +0 -1
  43. package/dist/chunk-HHWE3POT.js +0 -94
  44. package/dist/chunk-HHWE3POT.js.map +0 -1
  45. package/dist/chunk-HKUCJ437.js +0 -787
  46. package/dist/chunk-HKUCJ437.js.map +0 -1
  47. package/dist/chunk-JHCHEVET.js +0 -274
  48. package/dist/chunk-JHCHEVET.js.map +0 -1
  49. package/dist/chunk-JHOJHHU7.js +0 -867
  50. package/dist/chunk-JHOJHHU7.js.map +0 -1
  51. package/dist/chunk-JM2SKQMS.js +0 -750
  52. package/dist/chunk-JM2SKQMS.js.map +0 -1
  53. package/dist/chunk-JN2FCO5W.js +0 -7958
  54. package/dist/chunk-JN2FCO5W.js.map +0 -1
  55. package/dist/chunk-K4DBDHLK.js +0 -158
  56. package/dist/chunk-K4DBDHLK.js.map +0 -1
  57. package/dist/chunk-K6N6XJJX.js +0 -306
  58. package/dist/chunk-K6N6XJJX.js.map +0 -1
  59. package/dist/chunk-MA6HLL3S.js +0 -65
  60. package/dist/chunk-MA6HLL3S.js.map +0 -1
  61. package/dist/chunk-MAZ26DC7.js +0 -99
  62. package/dist/chunk-MAZ26DC7.js.map +0 -1
  63. package/dist/chunk-MOXWMGPC.js +0 -577
  64. package/dist/chunk-MOXWMGPC.js.map +0 -1
  65. package/dist/chunk-NJC7U437.js +0 -626
  66. package/dist/chunk-NJC7U437.js.map +0 -1
  67. package/dist/chunk-NPCTHQIO.js +0 -91
  68. package/dist/chunk-NPCTHQIO.js.map +0 -1
  69. package/dist/chunk-ONWEPEDO.js +0 -57
  70. package/dist/chunk-ONWEPEDO.js.map +0 -1
  71. package/dist/chunk-OYZAPX5G.js +0 -1526
  72. package/dist/chunk-OYZAPX5G.js.map +0 -1
  73. package/dist/chunk-PC4UYEBM.js +0 -166
  74. package/dist/chunk-PC4UYEBM.js.map +0 -1
  75. package/dist/chunk-PICTDURQ.js +0 -766
  76. package/dist/chunk-PICTDURQ.js.map +0 -1
  77. package/dist/chunk-PJQFMIOX.js +0 -1182
  78. package/dist/chunk-PJQFMIOX.js.map +0 -1
  79. package/dist/chunk-PXD6ZFNY.js +0 -1107
  80. package/dist/chunk-PXD6ZFNY.js.map +0 -1
  81. package/dist/chunk-PXE2VKMX.js +0 -140
  82. package/dist/chunk-PXE2VKMX.js.map +0 -1
  83. package/dist/chunk-PZ5AY32C.js +0 -10
  84. package/dist/chunk-PZ5AY32C.js.map +0 -1
  85. package/dist/chunk-QBRSJK47.js +0 -622
  86. package/dist/chunk-QBRSJK47.js.map +0 -1
  87. package/dist/chunk-QWMPPZ3X.js +0 -550
  88. package/dist/chunk-QWMPPZ3X.js.map +0 -1
  89. package/dist/chunk-S3UZOQ5Y.js +0 -328
  90. package/dist/chunk-S3UZOQ5Y.js.map +0 -1
  91. package/dist/chunk-S5TT5R3L.js +0 -2668
  92. package/dist/chunk-S5TT5R3L.js.map +0 -1
  93. package/dist/chunk-T4SQEITX.js +0 -95
  94. package/dist/chunk-T4SQEITX.js.map +0 -1
  95. package/dist/chunk-TT4KNT67.js +0 -124
  96. package/dist/chunk-TT4KNT67.js.map +0 -1
  97. package/dist/chunk-U5CHZ5M3.js +0 -357
  98. package/dist/chunk-U5CHZ5M3.js.map +0 -1
  99. package/dist/chunk-ULOKLHIQ.js +0 -1937
  100. package/dist/chunk-ULOKLHIQ.js.map +0 -1
  101. package/dist/chunk-VI2UW6B6.js +0 -162
  102. package/dist/chunk-VI2UW6B6.js.map +0 -1
  103. package/dist/chunk-VQMK5FMP.js +0 -247
  104. package/dist/chunk-VQMK5FMP.js.map +0 -1
  105. package/dist/chunk-VSMTAMNK.js +0 -53
  106. package/dist/chunk-VSMTAMNK.js.map +0 -1
  107. package/dist/chunk-VZSRQ272.js +0 -149
  108. package/dist/chunk-VZSRQ272.js.map +0 -1
  109. package/dist/chunk-WW2A73HW.js +0 -159
  110. package/dist/chunk-WW2A73HW.js.map +0 -1
  111. package/dist/chunk-X4UCIOTZ.js +0 -136
  112. package/dist/chunk-X4UCIOTZ.js.map +0 -1
  113. package/dist/chunk-XDIRG3TO.js +0 -1266
  114. package/dist/chunk-XDIRG3TO.js.map +0 -1
  115. package/dist/chunk-XJYR7XFV.js +0 -317
  116. package/dist/chunk-XJYR7XFV.js.map +0 -1
  117. package/dist/chunk-ZET2UAYW.js +0 -89
  118. package/dist/chunk-ZET2UAYW.js.map +0 -1
  119. package/dist/chunk-ZZUXHH3R.js +0 -99
  120. package/dist/chunk-ZZUXHH3R.js.map +0 -1
  121. package/dist/cli.d.ts +0 -1
  122. package/dist/cli.js +0 -112
  123. package/dist/cli.js.map +0 -1
  124. package/dist/contract/index.d.ts +0 -4969
  125. package/dist/contract/index.js +0 -1653
  126. package/dist/contract/index.js.map +0 -1
  127. package/dist/control.d.ts +0 -1013
  128. package/dist/control.js +0 -34
  129. package/dist/control.js.map +0 -1
  130. package/dist/fuzz.d.ts +0 -759
  131. package/dist/fuzz.js +0 -714
  132. package/dist/fuzz.js.map +0 -1
  133. package/dist/hosted/index.d.ts +0 -730
  134. package/dist/hosted/index.js +0 -14
  135. package/dist/hosted/index.js.map +0 -1
  136. package/dist/index.d.ts +0 -16780
  137. package/dist/index.js +0 -12168
  138. package/dist/index.js.map +0 -1
  139. package/dist/matrix/index.d.ts +0 -155
  140. package/dist/matrix/index.js +0 -8
  141. package/dist/matrix/index.js.map +0 -1
  142. package/dist/meta-eval/index.d.ts +0 -1030
  143. package/dist/meta-eval/index.js +0 -417
  144. package/dist/meta-eval/index.js.map +0 -1
  145. package/dist/multishot/index.d.ts +0 -579
  146. package/dist/multishot/index.js +0 -589
  147. package/dist/multishot/index.js.map +0 -1
  148. package/dist/openapi.json +0 -992
  149. package/dist/pipelines/index.d.ts +0 -567
  150. package/dist/pipelines/index.js +0 -515
  151. package/dist/pipelines/index.js.map +0 -1
  152. package/dist/reporting.d.ts +0 -1277
  153. package/dist/reporting.js +0 -48
  154. package/dist/reporting.js.map +0 -1
  155. package/dist/rl.d.ts +0 -4092
  156. package/dist/rl.js +0 -1724
  157. package/dist/rl.js.map +0 -1
  158. package/dist/run-campaign-HNFPJET4.js +0 -14
  159. package/dist/run-campaign-HNFPJET4.js.map +0 -1
  160. package/dist/storyboard/index.d.ts +0 -279
  161. package/dist/storyboard/index.js +0 -767
  162. package/dist/storyboard/index.js.map +0 -1
  163. package/dist/trace-attributes.d.ts +0 -52
  164. package/dist/trace-attributes.js +0 -62
  165. package/dist/trace-attributes.js.map +0 -1
  166. package/dist/traces.d.ts +0 -2343
  167. package/dist/traces.js +0 -249
  168. package/dist/traces.js.map +0 -1
  169. package/dist/wire/index.d.ts +0 -1252
  170. package/dist/wire/index.js +0 -81
  171. package/dist/wire/index.js.map +0 -1
@@ -1,766 +0,0 @@
1
- import {
2
- fsCampaignStorage,
3
- runCampaign
4
- } from "./chunk-XDIRG3TO.js";
5
- import {
6
- __export
7
- } from "./chunk-PZ5AY32C.js";
8
-
9
- // src/benchmarks/index.ts
10
- var benchmarks_exports = {};
11
- __export(benchmarks_exports, {
12
- BENCHMARK_SPLIT_SEED: () => BENCHMARK_SPLIT_SEED,
13
- buildStandardRetrievalItems: () => buildStandardRetrievalItems,
14
- calibrateBenchmarkMetric: () => calibrateBenchmarkMetric,
15
- createRetrievalIdBenchmarkAdapter: () => createRetrievalIdBenchmarkAdapter,
16
- deterministicSplit: () => deterministicSplit,
17
- evaluateStandardRetrieval: () => evaluateStandardRetrieval,
18
- normalizeRetrievedDocumentIds: () => normalizeRetrievedDocumentIds,
19
- parseBeirCorpusJsonl: () => parseBeirCorpusJsonl,
20
- parseBeirQueriesJsonl: () => parseBeirQueriesJsonl,
21
- parseJsonlRows: () => parseJsonlRows,
22
- parseQrels: () => parseQrels,
23
- parseTsvRows: () => parseTsvRows,
24
- renderBenchmarkReportMarkdown: () => renderBenchmarkReportMarkdown,
25
- retrievalMetricsAtCutoff: () => retrievalMetricsAtCutoff,
26
- routing: () => routing_exports,
27
- runBenchmarkAdapter: () => runBenchmarkAdapter,
28
- summarizeBenchmarkCampaign: () => summarizeBenchmarkCampaign
29
- });
30
-
31
- // src/benchmarks/calibration.ts
32
- async function calibrateBenchmarkMetric(options) {
33
- const weak = await options.adapter.evaluate(options.item, options.weakArtifact);
34
- const strong = await options.adapter.evaluate(options.item, options.strongArtifact);
35
- const weakScore = clamp01(weak.score);
36
- const strongScore = clamp01(strong.score);
37
- const maxWeakScore = options.maxWeakScore ?? 0.3;
38
- const minStrongScore = options.minStrongScore ?? 0.7;
39
- const minGap = options.minGap ?? 0.4;
40
- const gap = strongScore - weakScore;
41
- const reasons = [
42
- weakScore <= maxWeakScore ? void 0 : `weak score ${weakScore.toFixed(3)} exceeds max ${maxWeakScore.toFixed(3)}`,
43
- strongScore >= minStrongScore ? void 0 : `strong score ${strongScore.toFixed(3)} below min ${minStrongScore.toFixed(3)}`,
44
- gap >= minGap ? void 0 : `gap ${gap.toFixed(3)} below min ${minGap.toFixed(3)}`
45
- ].filter((reason) => Boolean(reason));
46
- return {
47
- passed: reasons.length === 0,
48
- weak,
49
- strong,
50
- weakScore,
51
- strongScore,
52
- gap,
53
- reasons
54
- };
55
- }
56
- function clamp01(value) {
57
- if (!Number.isFinite(value)) return 0;
58
- if (value < 0) return 0;
59
- if (value > 1) return 1;
60
- return value;
61
- }
62
-
63
- // src/benchmarks/routing/index.ts
64
- var routing_exports = {};
65
- __export(routing_exports, {
66
- ROUTING_DATASET: () => ROUTING_DATASET,
67
- RoutingAdapter: () => RoutingAdapter,
68
- assignSplit: () => assignSplit,
69
- evaluate: () => evaluate,
70
- extractRouteTokens: () => extractRouteTokens,
71
- loadDataset: () => loadDataset
72
- });
73
-
74
- // src/benchmarks/types.ts
75
- function fnv1a32(input) {
76
- let h = 2166136261;
77
- for (let i = 0; i < input.length; i++) {
78
- h ^= input.charCodeAt(i) & 255;
79
- h = h + ((h << 1) + (h << 4) + (h << 7) + (h << 8) + (h << 24)) >>> 0;
80
- }
81
- return h >>> 0;
82
- }
83
- var BENCHMARK_SPLIT_SEED = "agent-eval-v1";
84
- function deterministicSplit(itemId, seed = BENCHMARK_SPLIT_SEED) {
85
- const h = fnv1a32(`${seed}::${itemId}`);
86
- const pos = h / 4294967296;
87
- if (pos < 0.6) return "search";
88
- if (pos < 0.8) return "dev";
89
- return "holdout";
90
- }
91
-
92
- // src/benchmarks/routing/dataset.ts
93
- var ROUTING_DATASET = [
94
- {
95
- id: "file_001",
96
- category: "file",
97
- prompt: "Save the meeting notes to /tmp/notes-2025-04.md as markdown.",
98
- route: "fs.write",
99
- synonyms: ["filesystem.write", "write_file"],
100
- hardNegatives: ["fs.read", "chat.reply"]
101
- },
102
- {
103
- id: "file_002",
104
- category: "file",
105
- prompt: "Read the contents of /etc/hosts and summarize the entries.",
106
- route: "fs.read",
107
- synonyms: ["filesystem.read", "read_file"],
108
- hardNegatives: ["fs.write", "search.web"]
109
- },
110
- {
111
- id: "file_003",
112
- category: "file",
113
- prompt: "List every Python file under src/ recursively.",
114
- route: "fs.list",
115
- synonyms: ["filesystem.list", "list_files"],
116
- hardNegatives: ["fs.read", "search.code"]
117
- },
118
- {
119
- id: "file_004",
120
- category: "file",
121
- prompt: "Delete the cached build at .turbo/cache.",
122
- route: "fs.delete",
123
- synonyms: ["filesystem.delete", "remove_file"],
124
- hardNegatives: ["fs.write", "fs.list"]
125
- },
126
- {
127
- id: "math_001",
128
- category: "math",
129
- prompt: "What is the integral of 3x^2 + 2x from 0 to 5?",
130
- route: "math.integral",
131
- synonyms: ["calculator.integral", "math.solve"],
132
- hardNegatives: ["math.derivative", "chat.reply"]
133
- },
134
- {
135
- id: "math_002",
136
- category: "math",
137
- prompt: "Compute the derivative of sin(x) * cos(x).",
138
- route: "math.derivative",
139
- synonyms: ["calculator.derivative", "math.solve"],
140
- hardNegatives: ["math.integral", "math.algebra"]
141
- },
142
- {
143
- id: "math_003",
144
- category: "math",
145
- prompt: "Solve 2x + 7 = 19 for x.",
146
- route: "math.algebra",
147
- synonyms: ["calculator.algebra", "math.solve"],
148
- hardNegatives: ["math.derivative", "math.integral"]
149
- },
150
- {
151
- id: "math_004",
152
- category: "math",
153
- prompt: "What is the prime factorization of 360?",
154
- route: "math.numbertheory",
155
- synonyms: ["calculator.factor", "math.solve"],
156
- hardNegatives: ["math.algebra", "search.web"]
157
- },
158
- {
159
- id: "search_001",
160
- category: "search",
161
- prompt: "Find recent papers on agent prompt optimization with held-out promotion gates.",
162
- route: "search.web",
163
- synonyms: ["web.search", "search.papers"],
164
- hardNegatives: ["search.code", "chat.reply"]
165
- },
166
- {
167
- id: "search_002",
168
- category: "search",
169
- prompt: "Search the codebase for every call site of `runProposeReview`.",
170
- route: "search.code",
171
- synonyms: ["code.search", "grep"],
172
- hardNegatives: ["search.web", "fs.read"]
173
- },
174
- {
175
- id: "search_003",
176
- category: "search",
177
- prompt: "What is the latest release of the Tangle network on GitHub?",
178
- route: "search.web",
179
- synonyms: ["web.search", "github.releases"],
180
- hardNegatives: ["search.code", "chat.reply"]
181
- },
182
- {
183
- id: "search_004",
184
- category: "search",
185
- prompt: "Find all TODO comments in the agent-eval src tree.",
186
- route: "search.code",
187
- synonyms: ["code.search", "grep"],
188
- hardNegatives: ["search.web", "fs.list"]
189
- },
190
- {
191
- id: "chat_001",
192
- category: "chat",
193
- prompt: "Hi there, how are you doing today?",
194
- route: "chat.reply",
195
- synonyms: ["conversation.reply"],
196
- hardNegatives: ["search.web", "fs.read"]
197
- },
198
- {
199
- id: "chat_002",
200
- category: "chat",
201
- prompt: "Please explain the difference between an LLM and a foundation model.",
202
- route: "chat.reply",
203
- synonyms: ["conversation.reply", "qa.answer"],
204
- hardNegatives: ["search.web", "math.algebra"]
205
- },
206
- {
207
- id: "chat_003",
208
- category: "chat",
209
- prompt: "Tell me a short joke about distributed systems.",
210
- route: "chat.reply",
211
- synonyms: ["conversation.reply"],
212
- hardNegatives: ["search.web", "fs.read"]
213
- },
214
- {
215
- id: "chat_004",
216
- category: "chat",
217
- prompt: "Acknowledge my last message with a thumbs up.",
218
- route: "chat.reply",
219
- synonyms: ["conversation.reply", "react"],
220
- hardNegatives: ["fs.write", "search.web"]
221
- }
222
- ];
223
-
224
- // src/benchmarks/routing/index.ts
225
- var RoutingAdapter = class {
226
- id = "first-party/routing";
227
- family = "first-party";
228
- taskKind = "routing";
229
- description = "Synthetic fixed-route classification smoke benchmark";
230
- defaultMetric = "route_exact_match";
231
- async loadDataset(split) {
232
- return ROUTING_DATASET.map((item) => ({ id: item.id, payload: item })).filter(
233
- (it) => assignSplitImpl(it.id) === split
234
- );
235
- }
236
- async evaluate(item, response) {
237
- const tokens = extractRouteTokens(response);
238
- const correct = new Set(
239
- [item.payload.route, ...item.payload.synonyms].map((s) => s.toLowerCase())
240
- );
241
- const hardNeg = new Set(item.payload.hardNegatives.map((s) => s.toLowerCase()));
242
- const firstMatch = tokens.find((t) => correct.has(t.toLowerCase())) ?? null;
243
- const firstHardNeg = tokens.find((t) => hardNeg.has(t.toLowerCase())) ?? null;
244
- const score = firstMatch ? 1 : 0;
245
- return {
246
- score,
247
- raw: {
248
- firstToken: tokens[0] ?? null,
249
- matchedRoute: firstMatch,
250
- hitHardNegative: Boolean(firstHardNeg),
251
- hardNegativeRoute: firstHardNeg,
252
- category: item.payload.category
253
- }
254
- };
255
- }
256
- assignSplit(itemId) {
257
- return assignSplitImpl(itemId);
258
- }
259
- };
260
- function assignSplitImpl(itemId) {
261
- return deterministicSplit(`routing::${itemId}`);
262
- }
263
- function extractRouteTokens(response) {
264
- const matches = response.match(/[a-z][a-z0-9_]*\.[a-z][a-z0-9_]*/gi);
265
- return matches ?? [];
266
- }
267
- var adapter = new RoutingAdapter();
268
- var loadDataset = adapter.loadDataset.bind(adapter);
269
- var evaluate = adapter.evaluate.bind(adapter);
270
- var assignSplit = adapter.assignSplit.bind(adapter);
271
-
272
- // src/benchmarks/runner.ts
273
- import { join } from "path";
274
- async function runBenchmarkAdapter(options) {
275
- const storage = options.storage ?? fsCampaignStorage();
276
- const benchmarkId = benchmarkIdFor(options.adapter);
277
- const scenarios = await loadBenchmarkScenarios(options.adapter, options.splits);
278
- const judge = benchmarkAdapterJudge(options.adapter);
279
- const dispatch = async (scenario, context) => {
280
- return options.respond({ scenario, item: scenario.item, context });
281
- };
282
- const campaign = await runCampaign({
283
- scenarios,
284
- dispatch,
285
- dispatchRef: `benchmark:${benchmarkId}`,
286
- judges: [judge],
287
- seed: options.seed,
288
- reps: options.reps,
289
- resumable: options.resumable,
290
- costCeiling: options.costCeiling,
291
- maxConcurrency: options.maxConcurrency,
292
- dispatchTimeoutMs: options.dispatchTimeoutMs,
293
- expectUsage: options.expectUsage ?? "off",
294
- runDir: options.runDir,
295
- repo: options.repo,
296
- storage,
297
- now: options.now
298
- });
299
- const report = summarizeBenchmarkCampaign({
300
- adapter: options.adapter,
301
- scenarios,
302
- campaign
303
- });
304
- storage.ensureDir(campaign.runDir);
305
- const reportJsonPath = join(campaign.runDir, "benchmark-report.json");
306
- const reportMarkdownPath = join(campaign.runDir, "benchmark-report.md");
307
- storage.write(reportJsonPath, `${JSON.stringify(report, null, 2)}
308
- `);
309
- storage.write(reportMarkdownPath, renderBenchmarkReportMarkdown(report));
310
- return { scenarios, campaign, report, reportJsonPath, reportMarkdownPath };
311
- }
312
- async function loadBenchmarkScenarios(adapter2, splits = ["search", "dev", "holdout"]) {
313
- const benchmarkId = benchmarkIdFor(adapter2);
314
- const family = adapter2.family ?? "custom";
315
- const taskKind = adapter2.taskKind ?? "custom";
316
- const out = [];
317
- const seen = /* @__PURE__ */ new Set();
318
- for (const split of splits) {
319
- const items = await adapter2.loadDataset(split);
320
- for (const item of items) {
321
- const splitTag = item.split ?? adapter2.assignSplit(item.id);
322
- if (splitTag !== split) continue;
323
- const key = `${benchmarkId}:${item.id}`;
324
- if (seen.has(key)) continue;
325
- seen.add(key);
326
- out.push({
327
- id: key,
328
- kind: "benchmark",
329
- benchmarkId,
330
- family: item.family ?? family,
331
- taskKind: item.taskKind ?? taskKind,
332
- splitTag,
333
- tags: [.../* @__PURE__ */ new Set([splitTag, ...item.tags ?? []])],
334
- item
335
- });
336
- }
337
- }
338
- return out;
339
- }
340
- function benchmarkAdapterJudge(adapter2) {
341
- const benchmarkId = benchmarkIdFor(adapter2);
342
- return {
343
- name: `${benchmarkId}:score`,
344
- dimensions: [
345
- { key: "score", description: `Primary ${benchmarkId} benchmark score` },
346
- { key: "passed", description: "Binary pass projection for aggregate reporting" }
347
- ],
348
- appliesTo: (scenario) => {
349
- return scenario.kind === "benchmark" && scenario.id.startsWith(`${benchmarkId}:`);
350
- },
351
- async score({ artifact, scenario }) {
352
- const evaluation = await adapter2.evaluate(scenario.item, artifact);
353
- const dimensions = normalizeEvaluationDimensions(evaluation);
354
- return {
355
- dimensions,
356
- composite: clamp012(evaluation.score),
357
- notes: evaluation.notes ?? ""
358
- };
359
- }
360
- };
361
- }
362
- function summarizeBenchmarkCampaign(input) {
363
- const scenarioById = new Map(input.scenarios.map((scenario) => [scenario.id, scenario]));
364
- const rows = input.campaign.cells.map((cell) => {
365
- const scenario = scenarioById.get(cell.scenarioId);
366
- const judge = firstJudgeScore(cell.judgeScores);
367
- const score = judge?.composite ?? 0;
368
- return {
369
- cell,
370
- scenario,
371
- score,
372
- passed: (judge?.dimensions.passed ?? (score > 0 ? 1 : 0)) >= 1,
373
- dimensions: judge?.dimensions ?? {}
374
- };
375
- });
376
- const successful = rows.filter((row) => !row.cell.error);
377
- const scoreValues = successful.map((row) => row.score);
378
- return {
379
- benchmarkId: benchmarkIdFor(input.adapter),
380
- family: input.adapter.family ?? "custom",
381
- taskKind: input.adapter.taskKind ?? "custom",
382
- ...input.adapter.source ? { source: input.adapter.source } : {},
383
- runDir: input.campaign.runDir,
384
- manifestHash: input.campaign.manifestHash,
385
- seed: input.campaign.seed,
386
- startedAt: input.campaign.startedAt,
387
- endedAt: input.campaign.endedAt,
388
- durationMs: input.campaign.durationMs,
389
- totalItems: input.scenarios.length,
390
- totalCells: input.campaign.cells.length,
391
- cellsFailed: input.campaign.aggregates.cellsFailed,
392
- cellsCached: input.campaign.aggregates.cellsCached,
393
- totalCostUsd: input.campaign.aggregates.totalCostUsd,
394
- splits: summarizeSlices(successful, (row) => row.scenario?.splitTag ?? "unknown", [
395
- "search",
396
- "dev",
397
- "holdout"
398
- ]),
399
- tags: summarizeSlices(successful, (row) => row.scenario?.tags ?? []),
400
- dimensions: summarizeDimensions(successful.map((row) => row.dimensions)),
401
- score: distribution(scoreValues),
402
- costUsd: distribution(successful.map((row) => row.cell.costUsd)),
403
- latencyMs: distribution(successful.map((row) => row.cell.durationMs))
404
- };
405
- }
406
- function renderBenchmarkReportMarkdown(report) {
407
- const splitRows = Object.entries(report.splits).map(([split, summary]) => {
408
- return `| ${split} | ${summary.n} | ${fmt(summary.meanScore)} | ${fmt(summary.passRate)} | ${fmt(summary.score.p90)} | ${fmt(summary.costUsd.mean)} | ${fmt(summary.latencyMs.p90)} |`;
409
- }).join("\n");
410
- const dimRows = Object.entries(report.dimensions).sort(([a], [b]) => a.localeCompare(b)).map(([key, dist]) => `| ${key} | ${dist.n} | ${fmt(dist.mean)} | ${fmt(dist.p90)} |`).join("\n");
411
- return [
412
- `# Benchmark Report: ${report.benchmarkId}`,
413
- "",
414
- `- family: ${report.family}`,
415
- `- task kind: ${report.taskKind}`,
416
- `- run dir: ${report.runDir}`,
417
- `- manifest: ${report.manifestHash}`,
418
- `- cells: ${report.totalCells} total, ${report.cellsFailed} failed, ${report.cellsCached} cached`,
419
- `- cost: $${fmt(report.totalCostUsd)}`,
420
- `- score: mean ${fmt(report.score.mean)}, median ${fmt(report.score.median)}, p90 ${fmt(report.score.p90)}, n=${report.score.n}`,
421
- "",
422
- "## Splits",
423
- "",
424
- "| split | n | mean score | pass rate | score p90 | mean cost | latency p90 ms |",
425
- "| --- | ---: | ---: | ---: | ---: | ---: | ---: |",
426
- splitRows || "| none | 0 | 0 | 0 | 0 | 0 | 0 |",
427
- "",
428
- "## Dimensions",
429
- "",
430
- "| dimension | n | mean | p90 |",
431
- "| --- | ---: | ---: | ---: |",
432
- dimRows || "| none | 0 | 0 | 0 |",
433
- ""
434
- ].join("\n");
435
- }
436
- function benchmarkIdFor(adapter2) {
437
- return adapter2.id ?? `${adapter2.family ?? "custom"}/${adapter2.taskKind ?? "custom"}`;
438
- }
439
- function normalizeEvaluationDimensions(evaluation) {
440
- const dimensions = { score: clamp012(evaluation.score) };
441
- for (const [key, value] of Object.entries(evaluation.dimensions ?? {})) {
442
- if (Number.isFinite(value)) dimensions[key] = value;
443
- }
444
- dimensions.passed = evaluation.passed ?? evaluation.score > 0 ? 1 : 0;
445
- return dimensions;
446
- }
447
- function summarizeDimensions(rows) {
448
- const values = /* @__PURE__ */ new Map();
449
- for (const row of rows) {
450
- for (const [key, value] of Object.entries(row)) {
451
- if (!Number.isFinite(value)) continue;
452
- const list = values.get(key) ?? [];
453
- list.push(value);
454
- values.set(key, list);
455
- }
456
- }
457
- return Object.fromEntries([...values.entries()].map(([key, vals]) => [key, distribution(vals)]));
458
- }
459
- function summarizeSlices(rows, keyOf, knownKeys = []) {
460
- const grouped = /* @__PURE__ */ new Map();
461
- for (const key of knownKeys) grouped.set(key, []);
462
- for (const row of rows) {
463
- const keys = keyOf(row);
464
- for (const key of Array.isArray(keys) ? keys : [keys]) {
465
- const list = grouped.get(key) ?? [];
466
- list.push(row);
467
- grouped.set(key, list);
468
- }
469
- }
470
- const out = {};
471
- for (const [key, list] of grouped) {
472
- const withShape = list;
473
- out[key] = {
474
- n: list.length,
475
- meanScore: mean(withShape.map((row) => row.score)),
476
- passRate: mean(withShape.map((row) => row.passed ? 1 : 0)),
477
- score: distribution(withShape.map((row) => row.score)),
478
- costUsd: distribution(withShape.map((row) => row.cell.costUsd)),
479
- latencyMs: distribution(withShape.map((row) => row.cell.durationMs))
480
- };
481
- }
482
- return out;
483
- }
484
- function firstJudgeScore(judgeScores) {
485
- return Object.values(judgeScores)[0];
486
- }
487
- function distribution(values) {
488
- const finite = [...values].filter(Number.isFinite).sort((a, b) => a - b);
489
- if (finite.length === 0) return { n: 0, min: 0, mean: 0, median: 0, p90: 0, max: 0 };
490
- return {
491
- n: finite.length,
492
- min: finite[0],
493
- mean: mean(finite),
494
- median: percentile(finite, 0.5),
495
- p90: percentile(finite, 0.9),
496
- max: finite[finite.length - 1]
497
- };
498
- }
499
- function percentile(sortedValues, p) {
500
- if (sortedValues.length === 0) return 0;
501
- const index = Math.min(
502
- sortedValues.length - 1,
503
- Math.max(0, Math.ceil(p * sortedValues.length) - 1)
504
- );
505
- return sortedValues[index];
506
- }
507
- function mean(values) {
508
- const finite = values.filter(Number.isFinite);
509
- if (finite.length === 0) return 0;
510
- return finite.reduce((sum, value) => sum + value, 0) / finite.length;
511
- }
512
- function clamp012(value) {
513
- if (!Number.isFinite(value)) return 0;
514
- if (value < 0) return 0;
515
- if (value > 1) return 1;
516
- return value;
517
- }
518
- function fmt(value) {
519
- if (!Number.isFinite(value)) return "0";
520
- return value.toFixed(value === 0 || Math.abs(value) >= 10 ? 0 : 3);
521
- }
522
-
523
- // src/benchmarks/standard-formats.ts
524
- function parseJsonlRows(text) {
525
- return text.split(/\r?\n/).map((line) => line.trim()).filter(Boolean).map((line, index) => {
526
- try {
527
- return JSON.parse(line);
528
- } catch (error) {
529
- throw new Error(`invalid JSONL row ${index + 1}: ${error.message}`);
530
- }
531
- });
532
- }
533
- function parseTsvRows(text) {
534
- return text.split(/\r?\n/).map((line) => line.trim()).filter(Boolean).filter((line) => !line.startsWith("#")).map((line) => line.split(/\t|\s+/));
535
- }
536
- function parseQrels(text) {
537
- return parseTsvRows(text).flatMap((parts, index) => {
538
- if (parts.length < 3) return [];
539
- const [queryId, maybeZeroOrDocId, maybeDocIdOrScore, maybeScore] = parts;
540
- if (!queryId || !maybeZeroOrDocId || !maybeDocIdOrScore) return [];
541
- if (queryId.toLowerCase() === "query-id" || queryId.toLowerCase() === "qid") return [];
542
- const documentId = maybeScore === void 0 ? maybeZeroOrDocId : maybeDocIdOrScore;
543
- const scoreText = maybeScore === void 0 ? maybeDocIdOrScore : maybeScore;
544
- const score = Number(scoreText);
545
- if (!documentId || !Number.isFinite(score)) {
546
- throw new Error(`invalid qrels row ${index + 1}: expected query id, doc id, score`);
547
- }
548
- return [{ queryId, documentId, score }];
549
- });
550
- }
551
- function parseBeirCorpusJsonl(text) {
552
- return parseJsonlRows(text).map((row, index) => {
553
- const id = stringField(row, "_id") ?? stringField(row, "id");
554
- const body = stringField(row, "text") ?? stringField(row, "contents");
555
- if (!id || body === void 0) {
556
- throw new Error(`invalid BEIR corpus row ${index + 1}: expected _id/id and text/contents`);
557
- }
558
- return {
559
- id,
560
- title: stringField(row, "title"),
561
- text: body,
562
- metadata: stripKnown(row, ["_id", "id", "title", "text", "contents"])
563
- };
564
- });
565
- }
566
- function parseBeirQueriesJsonl(text) {
567
- return parseJsonlRows(text).map((row, index) => {
568
- const id = stringField(row, "_id") ?? stringField(row, "id") ?? stringField(row, "query_id");
569
- const query = stringField(row, "text") ?? stringField(row, "query");
570
- if (!id || !query) {
571
- throw new Error(`invalid BEIR query row ${index + 1}: expected _id/id and text/query`);
572
- }
573
- return {
574
- id,
575
- text: query,
576
- metadata: stripKnown(row, ["_id", "id", "query_id", "text", "query"])
577
- };
578
- });
579
- }
580
- function buildStandardRetrievalItems(options) {
581
- const qrelsByQuery = /* @__PURE__ */ new Map();
582
- for (const qrel of options.qrels) {
583
- if (qrel.score <= 0) continue;
584
- const list = qrelsByQuery.get(qrel.queryId) ?? [];
585
- list.push(qrel);
586
- qrelsByQuery.set(qrel.queryId, list);
587
- }
588
- const corpus = options.includeCorpusInPayload && options.corpus ? Object.fromEntries(options.corpus.map((document) => [document.id, document])) : void 0;
589
- return options.queries.flatMap((query) => {
590
- const qrels = qrelsByQuery.get(query.id) ?? [];
591
- if (qrels.length === 0) return [];
592
- const split = options.splitOf?.(query.id) ?? deterministicSplit(`${options.benchmarkId}:${query.id}`);
593
- return [
594
- {
595
- id: query.id,
596
- split,
597
- family: options.family,
598
- taskKind: "retrieval",
599
- tags: [.../* @__PURE__ */ new Set([...options.tags ?? [], split])],
600
- ...options.source ? { source: options.source } : {},
601
- ...query.metadata ? { metadata: query.metadata } : {},
602
- payload: {
603
- queryId: query.id,
604
- query: query.text,
605
- expectedDocumentIds: qrels.map((qrel) => qrel.documentId),
606
- expectedScores: Object.fromEntries(qrels.map((qrel) => [qrel.documentId, qrel.score])),
607
- ...corpus ? { corpus } : {},
608
- ...query.metadata ? { metadata: query.metadata } : {}
609
- }
610
- }
611
- ];
612
- });
613
- }
614
- function createRetrievalIdBenchmarkAdapter(options) {
615
- const items = buildStandardRetrievalItems(options);
616
- return {
617
- id: options.benchmarkId,
618
- family: options.family,
619
- taskKind: "retrieval",
620
- source: options.source,
621
- defaultMetric: options.primaryMetric ?? "ndcg@10",
622
- async loadDataset(split) {
623
- return items.filter((item) => item.split === split);
624
- },
625
- async evaluate(item, artifact) {
626
- return evaluateStandardRetrieval(item.payload, artifact, options);
627
- },
628
- assignSplit(itemId) {
629
- return options.splitOf?.(itemId) ?? deterministicSplit(`${options.benchmarkId}:${itemId}`);
630
- }
631
- };
632
- }
633
- function evaluateStandardRetrieval(payload, artifact, options = {}) {
634
- const rankedDocumentIds = normalizeRetrievedDocumentIds(
635
- artifact,
636
- options.responseIdPattern ?? /[A-Za-z0-9_.:/-]+/g
637
- );
638
- const cutoffs = normalizeCutoffs(options.cutoffs ?? [1, 3, 5, 10]);
639
- const dimensions = {
640
- expected_count: payload.expectedDocumentIds.length,
641
- returned_count: rankedDocumentIds.length
642
- };
643
- for (const cutoff of cutoffs) {
644
- Object.assign(
645
- dimensions,
646
- retrievalMetricsAtCutoff({
647
- rankedDocumentIds,
648
- expectedScores: payload.expectedScores,
649
- cutoff
650
- })
651
- );
652
- }
653
- const primaryMetric = options.primaryMetric ?? "ndcg@10";
654
- const passMetric = options.passMetric ?? "hit@10";
655
- const score = dimensions[primaryMetric] ?? dimensions[`ndcg@${cutoffs[cutoffs.length - 1]}`] ?? 0;
656
- const passScore = dimensions[passMetric] ?? dimensions[`hit@${cutoffs[cutoffs.length - 1]}`] ?? score;
657
- return {
658
- score,
659
- passed: passScore >= (options.passThreshold ?? 1),
660
- dimensions,
661
- raw: {
662
- rankedDocumentIds,
663
- expectedDocumentIds: payload.expectedDocumentIds,
664
- expectedScores: payload.expectedScores
665
- }
666
- };
667
- }
668
- function normalizeRetrievedDocumentIds(artifact, responseIdPattern = /[A-Za-z0-9_.:/-]+/g) {
669
- if (typeof artifact === "string") {
670
- return uniqueOrdered((artifact.match(responseIdPattern) ?? []).map((value) => value.trim()));
671
- }
672
- if (Array.isArray(artifact)) {
673
- const entries = artifact;
674
- return uniqueOrdered(
675
- entries.flatMap((entry) => {
676
- if (typeof entry === "string") return [entry];
677
- return [entry.documentId ?? entry.docId ?? entry.id ?? ""];
678
- })
679
- );
680
- }
681
- const objectArtifact = artifact;
682
- return uniqueOrdered(
683
- [
684
- ...objectArtifact.documentIds ?? [],
685
- ...objectArtifact.ids ?? [],
686
- ...(objectArtifact.results ?? []).map(
687
- (entry) => entry.documentId ?? entry.docId ?? entry.id ?? ""
688
- )
689
- ].map((value) => value.trim())
690
- );
691
- }
692
- function retrievalMetricsAtCutoff(input) {
693
- const top = input.rankedDocumentIds.slice(0, input.cutoff);
694
- const relevantIds = Object.keys(input.expectedScores).filter(
695
- (id) => input.expectedScores[id] > 0
696
- );
697
- const relevant = new Set(relevantIds);
698
- const hits = top.filter((id) => relevant.has(id)).length;
699
- const firstRelevantRank = top.findIndex((id) => relevant.has(id));
700
- const prefix = `@${input.cutoff}`;
701
- return {
702
- [`hit${prefix}`]: hits > 0 ? 1 : 0,
703
- [`recall${prefix}`]: relevant.size === 0 ? 1 : hits / relevant.size,
704
- [`precision${prefix}`]: hits / input.cutoff,
705
- [`mrr${prefix}`]: firstRelevantRank === -1 ? 0 : 1 / (firstRelevantRank + 1),
706
- [`ndcg${prefix}`]: ndcgAt(input.rankedDocumentIds, input.expectedScores, input.cutoff)
707
- };
708
- }
709
- function stringField(row, key) {
710
- const value = row[key];
711
- return typeof value === "string" ? value : void 0;
712
- }
713
- function stripKnown(row, keys) {
714
- const known = new Set(keys);
715
- return Object.fromEntries(Object.entries(row).filter(([key]) => !known.has(key)));
716
- }
717
- function normalizeCutoffs(cutoffs) {
718
- return [
719
- ...new Set(cutoffs.map((cutoff) => Math.trunc(cutoff)).filter((cutoff) => cutoff > 0))
720
- ].sort((a, b) => a - b);
721
- }
722
- function uniqueOrdered(values) {
723
- const seen = /* @__PURE__ */ new Set();
724
- const out = [];
725
- for (const value of values) {
726
- const trimmed = value.trim();
727
- if (!trimmed || seen.has(trimmed)) continue;
728
- seen.add(trimmed);
729
- out.push(trimmed);
730
- }
731
- return out;
732
- }
733
- function ndcgAt(rankedDocumentIds, expectedScores, cutoff) {
734
- const dcg = rankedDocumentIds.slice(0, cutoff).reduce(
735
- (sum, documentId, index) => sum + discountedGain(expectedScores[documentId] ?? 0, index),
736
- 0
737
- );
738
- const ideal = Object.values(expectedScores).filter((score) => score > 0).sort((a, b) => b - a).slice(0, cutoff).reduce((sum, score, index) => sum + discountedGain(score, index), 0);
739
- return ideal === 0 ? 1 : dcg / ideal;
740
- }
741
- function discountedGain(relevance, zeroBasedRank) {
742
- if (relevance <= 0) return 0;
743
- return (2 ** relevance - 1) / Math.log2(zeroBasedRank + 2);
744
- }
745
-
746
- export {
747
- calibrateBenchmarkMetric,
748
- BENCHMARK_SPLIT_SEED,
749
- deterministicSplit,
750
- routing_exports,
751
- runBenchmarkAdapter,
752
- summarizeBenchmarkCampaign,
753
- renderBenchmarkReportMarkdown,
754
- parseJsonlRows,
755
- parseTsvRows,
756
- parseQrels,
757
- parseBeirCorpusJsonl,
758
- parseBeirQueriesJsonl,
759
- buildStandardRetrievalItems,
760
- createRetrievalIdBenchmarkAdapter,
761
- evaluateStandardRetrieval,
762
- normalizeRetrievedDocumentIds,
763
- retrievalMetricsAtCutoff,
764
- benchmarks_exports
765
- };
766
- //# sourceMappingURL=chunk-PICTDURQ.js.map