@tangle-network/agent-eval 0.117.1 → 0.118.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (143) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/dist/analyst/index.d.ts +2772 -21
  3. package/dist/analyst/index.js +7 -6
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/belief-state/index.d.ts +706 -10
  6. package/dist/belief-state/index.js +2 -1
  7. package/dist/belief-state/index.js.map +1 -1
  8. package/dist/benchmarks/index.d.ts +958 -14
  9. package/dist/benchmarks/index.js +12 -10
  10. package/dist/builder-eval/index.d.ts +449 -4
  11. package/dist/builder-eval/index.js +4 -3
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/campaign/index.d.ts +4275 -73
  14. package/dist/campaign/index.js +12 -10
  15. package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
  16. package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
  17. package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
  18. package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
  19. package/dist/chunk-FTUMG2U7.js.map +1 -0
  20. package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
  21. package/dist/chunk-HKUCJ437.js.map +1 -0
  22. package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
  23. package/dist/chunk-KSDQVPLR.js +286 -0
  24. package/dist/chunk-KSDQVPLR.js.map +1 -0
  25. package/dist/chunk-MA6HLL3S.js +65 -0
  26. package/dist/chunk-MA6HLL3S.js.map +1 -0
  27. package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
  28. package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
  29. package/dist/chunk-PXE2VKMX.js +140 -0
  30. package/dist/chunk-PXE2VKMX.js.map +1 -0
  31. package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
  32. package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
  33. package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
  34. package/dist/chunk-QKEGNI5B.js.map +1 -0
  35. package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
  36. package/dist/chunk-S3UZOQ5Y.js.map +1 -0
  37. package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
  38. package/dist/chunk-SVH2ANFD.js.map +1 -0
  39. package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
  40. package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
  41. package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
  42. package/dist/chunk-WDHBCA3M.js +31 -0
  43. package/dist/chunk-WDHBCA3M.js.map +1 -0
  44. package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
  45. package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
  46. package/dist/chunk-ZET2UAYW.js.map +1 -0
  47. package/dist/contract/index.d.ts +4012 -38
  48. package/dist/contract/index.js +56 -29
  49. package/dist/contract/index.js.map +1 -1
  50. package/dist/control.d.ts +1013 -9
  51. package/dist/control.js +4 -3
  52. package/dist/fuzz.d.ts +194 -4
  53. package/dist/fuzz.js +3 -3
  54. package/dist/hosted/index.d.ts +498 -17
  55. package/dist/index.d.ts +10992 -1288
  56. package/dist/index.js +101 -80
  57. package/dist/index.js.map +1 -1
  58. package/dist/matrix/index.d.ts +139 -4
  59. package/dist/meta-eval/index.d.ts +862 -15
  60. package/dist/meta-eval/index.js +2 -1
  61. package/dist/meta-eval/index.js.map +1 -1
  62. package/dist/multishot/index.d.ts +214 -14
  63. package/dist/openapi.json +1 -1
  64. package/dist/pipelines/index.d.ts +392 -7
  65. package/dist/pipelines/index.js +5 -3
  66. package/dist/pipelines/index.js.map +1 -1
  67. package/dist/reporting.d.ts +1277 -17
  68. package/dist/rl.d.ts +2359 -28
  69. package/dist/rl.js +6 -5
  70. package/dist/rl.js.map +1 -1
  71. package/dist/storyboard/index.d.ts +86 -1
  72. package/dist/trace-attributes.d.ts +16 -0
  73. package/dist/trace-attributes.js +32 -0
  74. package/dist/trace-attributes.js.map +1 -0
  75. package/dist/traces.d.ts +1978 -697
  76. package/dist/traces.js +53 -32
  77. package/dist/wire/index.d.ts +655 -9
  78. package/docs/insight-report.md +44 -0
  79. package/package.json +9 -3
  80. package/dist/adversarial-B7loGVVX.d.ts +0 -19
  81. package/dist/analyst-C8HHvfJp.d.ts +0 -88
  82. package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
  83. package/dist/baseline-DKq3gJpP.d.ts +0 -141
  84. package/dist/calibration-C8MTS7cw.d.ts +0 -101
  85. package/dist/chunk-5UF54T55.js.map +0 -1
  86. package/dist/chunk-E4BUPP7Z.js.map +0 -1
  87. package/dist/chunk-FQNLDL4D.js.map +0 -1
  88. package/dist/chunk-LQUTGLOZ.js.map +0 -1
  89. package/dist/chunk-S2F4J57L.js.map +0 -1
  90. package/dist/chunk-YZPO4UHR.js.map +0 -1
  91. package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
  92. package/dist/control-6vuGfmDH.d.ts +0 -258
  93. package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
  94. package/dist/dataset-NENEzRgk.d.ts +0 -115
  95. package/dist/default-registry-DaK8b3fv.d.ts +0 -155
  96. package/dist/emitter-CjD7vUwv.d.ts +0 -122
  97. package/dist/errors-oeQrLqXC.d.ts +0 -74
  98. package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
  99. package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
  100. package/dist/gepa-eESocoDi.d.ts +0 -642
  101. package/dist/index-PdX4VnPA.d.ts +0 -423
  102. package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
  103. package/dist/integrity-DqlBiLyK.d.ts +0 -81
  104. package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
  105. package/dist/kind-factory-ClZmO25A.d.ts +0 -171
  106. package/dist/llm-client-qoDd18Qz.d.ts +0 -289
  107. package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
  108. package/dist/off-policy-DiwuKKg7.d.ts +0 -132
  109. package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
  110. package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
  111. package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
  112. package/dist/provenance-DpjwyseI.d.ts +0 -541
  113. package/dist/query-CF7PG61p.d.ts +0 -35
  114. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  115. package/dist/release-report-C8G2i5Xi.d.ts +0 -236
  116. package/dist/researcher-C8XyxQsu.d.ts +0 -387
  117. package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
  118. package/dist/run-record-BDH49H2E.d.ts +0 -360
  119. package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
  120. package/dist/schema-B3Q3l9Z_.d.ts +0 -201
  121. package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
  122. package/dist/sequential-5iSVfzl2.d.ts +0 -139
  123. package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
  124. package/dist/statistics-KUnG73jH.d.ts +0 -494
  125. package/dist/storage-DrX3v_5B.d.ts +0 -50
  126. package/dist/store-C1YxJDEK.d.ts +0 -248
  127. package/dist/store-DGqD0Pyo.d.ts +0 -116
  128. package/dist/summary-report-C5bKFfm-.d.ts +0 -445
  129. package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
  130. package/dist/types-BSw1rOUB.d.ts +0 -634
  131. package/dist/types-BUxNaJ8c.d.ts +0 -108
  132. package/dist/types-BkfcQnxV.d.ts +0 -313
  133. package/dist/verdict-C9MlYujm.d.ts +0 -35
  134. /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
  135. /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
  136. /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
  137. /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
  138. /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
  139. /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
  140. /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
  141. /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
  142. /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
  143. /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
@@ -1,423 +0,0 @@
1
- import { S as Scenario, D as DispatchContext, C as CampaignResult } from './types-BSw1rOUB.js';
2
- import { a as RunSplitTag } from './run-record-BDH49H2E.js';
3
- import { C as CampaignStorage } from './storage-DrX3v_5B.js';
4
-
5
- /**
6
- * Shared types for the reference benchmark wrappers under
7
- * `src/benchmarks/`. Each wrapper exports the three functions in
8
- * `BenchmarkAdapter` plus its own typed `DatasetItem` shape.
9
- */
10
-
11
- type BenchmarkTaskKind = 'retrieval' | 'rag-answer' | 'hallucination' | 'kb-improvement' | 'routing' | 'custom';
12
- type BenchmarkFamily = 'beir' | 'mteb-retrieval' | 'msmarco' | 'trec-dl' | 'miracl' | 'lotte' | 'bright' | 'crag' | 'hotpotqa' | 'kilt' | 'ragtruth' | 'faithbench' | 'first-party' | 'custom';
13
- interface BenchmarkDatasetItem<TPayload = unknown> {
14
- /** Stable dataset-local item id (used for split assignment + paper
15
- * references). Unique within a benchmark. */
16
- id: string;
17
- /** Free-form payload. Each benchmark defines its own shape. */
18
- payload: TPayload;
19
- /** Optional precomputed split. When absent, adapters use assignSplit(id). */
20
- split?: RunSplitTag;
21
- /** Benchmark family this row came from, e.g. `beir` or `crag`. */
22
- family?: BenchmarkFamily | string;
23
- /** Benchmark-local task kind, e.g. retrieval vs answer quality. */
24
- taskKind?: BenchmarkTaskKind | string;
25
- /** Slice labels such as language, domain, freshness, multihop, long-tail. */
26
- tags?: string[];
27
- /** Dataset provenance, version, URL, or license notes. */
28
- source?: BenchmarkSource;
29
- metadata?: Record<string, unknown>;
30
- }
31
- interface BenchmarkEvaluation {
32
- /** [0, 1] score for the response on this item. Exact-match
33
- * benchmarks use 0/1; partial-credit benchmarks may return
34
- * fractional values. */
35
- score: number;
36
- /** Optional pass/fail projection. Defaults to `score > 0` when absent. */
37
- passed?: boolean;
38
- /** Numeric sub-metrics. These become report dimensions. */
39
- dimensions?: Record<string, number>;
40
- /** Optional bag of raw scoring signals — e.g. parsed numeric
41
- * answer, regex match, judge sub-scores. */
42
- raw: Record<string, unknown>;
43
- notes?: string;
44
- }
45
- interface BenchmarkSource {
46
- name?: string;
47
- url?: string;
48
- version?: string;
49
- license?: string;
50
- citation?: string;
51
- }
52
- /** Common signature implemented by every adapter under `src/benchmarks/*`. */
53
- interface BenchmarkAdapter<_TItem = unknown, TPayload = unknown, TArtifact = string> {
54
- /** Stable benchmark id such as `beir/nfcorpus` or `crag/smoke`. */
55
- id?: string;
56
- family?: BenchmarkFamily | string;
57
- taskKind?: BenchmarkTaskKind | string;
58
- description?: string;
59
- source?: BenchmarkSource;
60
- defaultMetric?: string;
61
- /** Load the dataset for the given split. May hit the network on
62
- * first call but should be cache-friendly. Adapters that don't
63
- * ship the dataset itself MUST throw a clearly-marked error
64
- * pointing the caller at the loader script. */
65
- loadDataset(split: RunSplitTag): Promise<BenchmarkDatasetItem<TPayload>[]>;
66
- /** Score a single response. Pure with respect to the inputs. */
67
- evaluate(item: BenchmarkDatasetItem<TPayload>, artifact: TArtifact): Promise<BenchmarkEvaluation>;
68
- /** Deterministic split assignment via item id hashing. The
69
- * fraction of items in each split is implementation-defined but
70
- * MUST be stable across processes and platforms. */
71
- assignSplit(itemId: string): RunSplitTag;
72
- }
73
- interface BenchmarkScenario<TPayload = unknown> extends Scenario {
74
- kind: 'benchmark';
75
- benchmarkId: string;
76
- family: BenchmarkFamily | string;
77
- taskKind: BenchmarkTaskKind | string;
78
- splitTag: RunSplitTag;
79
- item: BenchmarkDatasetItem<TPayload>;
80
- }
81
- type BenchmarkResponder<TPayload = unknown, TArtifact = string> = (input: {
82
- scenario: BenchmarkScenario<TPayload>;
83
- item: BenchmarkDatasetItem<TPayload>;
84
- context: DispatchContext;
85
- }) => Promise<TArtifact> | TArtifact;
86
- /** Split-assignment seed shared across all benchmarks. Bumping this
87
- * value reshuffles every split — do NOT do that lightly. */
88
- declare const BENCHMARK_SPLIT_SEED = "agent-eval-v1";
89
- /**
90
- * Assign an item id to one of `'search' | 'dev' | 'holdout'` using a
91
- * stable 32-bit hash of `${seed}::${id}`. Default proportions:
92
- *
93
- * search: 60% (optimization-readable)
94
- * dev: 20% (held-out for tuning, leak-on-purpose during dev)
95
- * holdout:20% (paper-grade held-out, gated reads)
96
- */
97
- declare function deterministicSplit(itemId: string, seed?: string): RunSplitTag;
98
-
99
- interface BenchmarkMetricCalibrationOptions<TPayload = unknown, TArtifact = string> {
100
- adapter: BenchmarkAdapter<BenchmarkDatasetItem<TPayload>, TPayload, TArtifact>;
101
- item: BenchmarkDatasetItem<TPayload>;
102
- weakArtifact: TArtifact;
103
- strongArtifact: TArtifact;
104
- maxWeakScore?: number;
105
- minStrongScore?: number;
106
- minGap?: number;
107
- }
108
- interface BenchmarkMetricCalibrationResult {
109
- passed: boolean;
110
- weak: BenchmarkEvaluation;
111
- strong: BenchmarkEvaluation;
112
- weakScore: number;
113
- strongScore: number;
114
- gap: number;
115
- reasons: string[];
116
- }
117
- declare function calibrateBenchmarkMetric<TPayload = unknown, TArtifact = string>(options: BenchmarkMetricCalibrationOptions<TPayload, TArtifact>): Promise<BenchmarkMetricCalibrationResult>;
118
-
119
- /**
120
- * Synthetic routing dataset. 16 tasks across 4 categories. Used as a
121
- * deterministic, dependency-free benchmark for any router that maps a
122
- * natural-language request to one of a fixed set of route labels.
123
- *
124
- * Format (see `routing/README.md` for prose):
125
- *
126
- * {
127
- * id: stable per-task ID (matches across processes).
128
- * category: one of the four route labels.
129
- * prompt: the user-facing request the router must classify.
130
- * route: the ground-truth route the router should pick.
131
- * synonyms: other strings that count as a correct answer.
132
- * hardNegatives:close-but-wrong route labels — used to detect the
133
- * "always picks the popular route" failure mode.
134
- * }
135
- *
136
- * The four categories are intentionally cross-domain (file ops,
137
- * math, search, conversation) so a router that collapses to one
138
- * category is easy to spot.
139
- */
140
- interface RoutingItem {
141
- id: string;
142
- category: 'file' | 'math' | 'search' | 'chat';
143
- prompt: string;
144
- /** Canonical correct route label. */
145
- route: string;
146
- /** Alternate route labels that also count as correct. */
147
- synonyms: string[];
148
- /** Wrong-but-tempting route labels (for analysis, not grading). */
149
- hardNegatives: string[];
150
- }
151
- declare const ROUTING_DATASET: RoutingItem[];
152
-
153
- /**
154
- * Routing benchmark — synthetic, dependency-free, ships in the
155
- * package. 16 cross-category items in `dataset.ts`. See
156
- * `routing/README.md` for the format.
157
- *
158
- * `evaluate` does case-insensitive exact match against the canonical
159
- * route plus declared synonyms. The first valid route token in the
160
- * response wins; everything else is ignored. Wrong answers also
161
- * report whether they hit a hard negative — useful when triaging
162
- * "always picks the popular route" failure modes.
163
- */
164
-
165
- type RoutingPayload = RoutingItem;
166
- type RoutingDatasetItem = BenchmarkDatasetItem<RoutingPayload>;
167
- declare class RoutingAdapter implements BenchmarkAdapter<RoutingDatasetItem, RoutingPayload> {
168
- readonly id = "first-party/routing";
169
- readonly family = "first-party";
170
- readonly taskKind = "routing";
171
- readonly description = "Synthetic fixed-route classification smoke benchmark";
172
- readonly defaultMetric = "route_exact_match";
173
- loadDataset(split: RunSplitTag): Promise<RoutingDatasetItem[]>;
174
- evaluate(item: RoutingDatasetItem, response: string): Promise<BenchmarkEvaluation>;
175
- assignSplit(itemId: string): RunSplitTag;
176
- }
177
- /**
178
- * Pull route-shaped tokens out of a model response. Routes look like
179
- * `category.action` (`fs.write`, `chat.reply`). Bare alphanumerics
180
- * are not routes, but `category.action` patterns are robust to most
181
- * model wrappers (JSON output, prose explanations, code fences).
182
- */
183
- declare function extractRouteTokens(response: string): string[];
184
- declare const loadDataset: (split: RunSplitTag) => Promise<RoutingDatasetItem[]>;
185
- declare const evaluate: (item: RoutingDatasetItem, response: string) => Promise<BenchmarkEvaluation>;
186
- declare const assignSplit: (itemId: string) => RunSplitTag;
187
-
188
- declare const index$1_ROUTING_DATASET: typeof ROUTING_DATASET;
189
- type index$1_RoutingAdapter = RoutingAdapter;
190
- declare const index$1_RoutingAdapter: typeof RoutingAdapter;
191
- type index$1_RoutingDatasetItem = RoutingDatasetItem;
192
- type index$1_RoutingItem = RoutingItem;
193
- type index$1_RoutingPayload = RoutingPayload;
194
- declare const index$1_assignSplit: typeof assignSplit;
195
- declare const index$1_evaluate: typeof evaluate;
196
- declare const index$1_extractRouteTokens: typeof extractRouteTokens;
197
- declare const index$1_loadDataset: typeof loadDataset;
198
- declare namespace index$1 {
199
- export { index$1_ROUTING_DATASET as ROUTING_DATASET, index$1_RoutingAdapter as RoutingAdapter, type index$1_RoutingDatasetItem as RoutingDatasetItem, type index$1_RoutingItem as RoutingItem, type index$1_RoutingPayload as RoutingPayload, index$1_assignSplit as assignSplit, index$1_evaluate as evaluate, index$1_extractRouteTokens as extractRouteTokens, index$1_loadDataset as loadDataset };
200
- }
201
-
202
- interface BenchmarkRunOptions<TPayload = unknown, TArtifact = string> {
203
- adapter: BenchmarkAdapter<BenchmarkDatasetItem<TPayload>, TPayload, TArtifact>;
204
- respond: BenchmarkResponder<TPayload, TArtifact>;
205
- splits?: readonly RunSplitTag[];
206
- runDir: string;
207
- repo?: string;
208
- seed?: number;
209
- reps?: number;
210
- resumable?: boolean;
211
- costCeiling?: number;
212
- maxConcurrency?: number;
213
- dispatchTimeoutMs?: number;
214
- expectUsage?: 'assert' | 'warn' | 'off';
215
- storage?: CampaignStorage;
216
- now?: () => Date;
217
- }
218
- interface BenchmarkReport {
219
- benchmarkId: string;
220
- family: BenchmarkFamily | string;
221
- taskKind: BenchmarkTaskKind | string;
222
- source?: BenchmarkAdapter['source'];
223
- runDir: string;
224
- manifestHash: string;
225
- seed: number;
226
- startedAt: string;
227
- endedAt: string;
228
- durationMs: number;
229
- totalItems: number;
230
- totalCells: number;
231
- cellsFailed: number;
232
- cellsCached: number;
233
- totalCostUsd: number;
234
- splits: Record<string, BenchmarkSliceSummary>;
235
- tags: Record<string, BenchmarkSliceSummary>;
236
- dimensions: Record<string, BenchmarkDistribution>;
237
- score: BenchmarkDistribution;
238
- costUsd: BenchmarkDistribution;
239
- latencyMs: BenchmarkDistribution;
240
- }
241
- interface BenchmarkSliceSummary {
242
- n: number;
243
- meanScore: number;
244
- passRate: number;
245
- score: BenchmarkDistribution;
246
- costUsd: BenchmarkDistribution;
247
- latencyMs: BenchmarkDistribution;
248
- }
249
- interface BenchmarkDistribution {
250
- n: number;
251
- min: number;
252
- mean: number;
253
- median: number;
254
- p90: number;
255
- max: number;
256
- }
257
- interface BenchmarkRunResult<TPayload = unknown, TArtifact = string> {
258
- scenarios: Array<BenchmarkScenario<TPayload>>;
259
- campaign: CampaignResult<TArtifact, BenchmarkScenario<TPayload>>;
260
- report: BenchmarkReport;
261
- reportJsonPath: string;
262
- reportMarkdownPath: string;
263
- }
264
- declare function runBenchmarkAdapter<TPayload = unknown, TArtifact = string>(options: BenchmarkRunOptions<TPayload, TArtifact>): Promise<BenchmarkRunResult<TPayload, TArtifact>>;
265
- declare function summarizeBenchmarkCampaign<TPayload, TArtifact>(input: {
266
- adapter: BenchmarkAdapter<BenchmarkDatasetItem<TPayload>, TPayload, TArtifact>;
267
- scenarios: Array<BenchmarkScenario<TPayload>>;
268
- campaign: CampaignResult<TArtifact, BenchmarkScenario<TPayload>>;
269
- }): BenchmarkReport;
270
- declare function renderBenchmarkReportMarkdown(report: BenchmarkReport): string;
271
-
272
- interface StandardRetrievalDocument {
273
- id: string;
274
- title?: string;
275
- text: string;
276
- metadata?: Record<string, unknown>;
277
- }
278
- interface StandardRetrievalQuery {
279
- id: string;
280
- text: string;
281
- metadata?: Record<string, unknown>;
282
- }
283
- interface StandardRetrievalQrel {
284
- queryId: string;
285
- documentId: string;
286
- score: number;
287
- }
288
- interface StandardRetrievalPayload {
289
- queryId: string;
290
- query: string;
291
- expectedDocumentIds: string[];
292
- expectedScores: Record<string, number>;
293
- corpus?: Record<string, StandardRetrievalDocument>;
294
- metadata?: Record<string, unknown>;
295
- }
296
- interface BuildStandardRetrievalItemsOptions {
297
- benchmarkId: string;
298
- family: BenchmarkFamily | string;
299
- queries: readonly StandardRetrievalQuery[];
300
- qrels: readonly StandardRetrievalQrel[];
301
- corpus?: readonly StandardRetrievalDocument[];
302
- includeCorpusInPayload?: boolean;
303
- source?: BenchmarkSource;
304
- tags?: readonly string[];
305
- splitOf?: (queryId: string) => RunSplitTag;
306
- }
307
- interface RetrievalIdAdapterOptions extends BuildStandardRetrievalItemsOptions {
308
- responseIdPattern?: RegExp;
309
- cutoffs?: readonly number[];
310
- primaryMetric?: string;
311
- passMetric?: string;
312
- passThreshold?: number;
313
- }
314
- interface StandardRetrievalResult {
315
- id?: string;
316
- documentId?: string;
317
- docId?: string;
318
- score?: number;
319
- }
320
- type StandardRetrievalArtifact = string | readonly string[] | readonly StandardRetrievalResult[] | {
321
- ids?: readonly string[];
322
- documentIds?: readonly string[];
323
- results?: readonly StandardRetrievalResult[];
324
- };
325
- interface StandardRetrievalEvaluationOptions {
326
- responseIdPattern?: RegExp;
327
- cutoffs?: readonly number[];
328
- primaryMetric?: string;
329
- passMetric?: string;
330
- passThreshold?: number;
331
- }
332
- declare function parseJsonlRows<T = unknown>(text: string): T[];
333
- declare function parseTsvRows(text: string): string[][];
334
- declare function parseQrels(text: string): StandardRetrievalQrel[];
335
- declare function parseBeirCorpusJsonl(text: string): StandardRetrievalDocument[];
336
- declare function parseBeirQueriesJsonl(text: string): StandardRetrievalQuery[];
337
- declare function buildStandardRetrievalItems(options: BuildStandardRetrievalItemsOptions): Array<BenchmarkDatasetItem<StandardRetrievalPayload>>;
338
- declare function createRetrievalIdBenchmarkAdapter(options: RetrievalIdAdapterOptions): BenchmarkAdapter<BenchmarkDatasetItem<StandardRetrievalPayload>, StandardRetrievalPayload, StandardRetrievalArtifact>;
339
- declare function evaluateStandardRetrieval(payload: StandardRetrievalPayload, artifact: StandardRetrievalArtifact, options?: StandardRetrievalEvaluationOptions): {
340
- score: number;
341
- passed: boolean;
342
- dimensions: Record<string, number>;
343
- raw: {
344
- rankedDocumentIds: string[];
345
- expectedDocumentIds: string[];
346
- expectedScores: Record<string, number>;
347
- };
348
- };
349
- declare function normalizeRetrievedDocumentIds(artifact: StandardRetrievalArtifact, responseIdPattern?: RegExp): string[];
350
- declare function retrievalMetricsAtCutoff(input: {
351
- rankedDocumentIds: readonly string[];
352
- expectedScores: Record<string, number>;
353
- cutoff: number;
354
- }): Record<string, number>;
355
-
356
- /**
357
- * Reference benchmark wrappers — entry point.
358
- *
359
- * Core surface (exported here):
360
- * - The `BenchmarkAdapter` contract.
361
- * - `runBenchmarkAdapter` for campaign-backed benchmark execution.
362
- * - `calibrateBenchmarkMetric` for weak/strong metric checks.
363
- * - Standard retrieval parsers for BEIR/MTEB/MS MARCO/TREC/MIRACL-style files.
364
- * - `deterministicSplit` + `BENCHMARK_SPLIT_SEED` for split assignment.
365
- * - `routing` — synthetic 16-task router benchmark. The only novel
366
- * benchmark we built; ships in the package.
367
- *
368
- * Example wrappers (under `examples/benchmarks/`, NOT in the bundle):
369
- * - `gsm8k` — exact-match math reasoning (HF mirror, dataset
370
- * not bundled).
371
- * - `swebench-lite` — 30-instance SWE-Bench subset via an external
372
- * grader command.
373
- *
374
- * The example wrappers are reference implementations of `BenchmarkAdapter`.
375
- * Read them, copy them, adapt them. They're intentionally not in the main
376
- * entry — every team will configure them differently.
377
- */
378
-
379
- declare const index_BENCHMARK_SPLIT_SEED: typeof BENCHMARK_SPLIT_SEED;
380
- type index_BenchmarkAdapter<_TItem = unknown, TPayload = unknown, TArtifact = string> = BenchmarkAdapter<_TItem, TPayload, TArtifact>;
381
- type index_BenchmarkDatasetItem<TPayload = unknown> = BenchmarkDatasetItem<TPayload>;
382
- type index_BenchmarkDistribution = BenchmarkDistribution;
383
- type index_BenchmarkEvaluation = BenchmarkEvaluation;
384
- type index_BenchmarkFamily = BenchmarkFamily;
385
- type index_BenchmarkMetricCalibrationOptions<TPayload = unknown, TArtifact = string> = BenchmarkMetricCalibrationOptions<TPayload, TArtifact>;
386
- type index_BenchmarkMetricCalibrationResult = BenchmarkMetricCalibrationResult;
387
- type index_BenchmarkReport = BenchmarkReport;
388
- type index_BenchmarkResponder<TPayload = unknown, TArtifact = string> = BenchmarkResponder<TPayload, TArtifact>;
389
- type index_BenchmarkRunOptions<TPayload = unknown, TArtifact = string> = BenchmarkRunOptions<TPayload, TArtifact>;
390
- type index_BenchmarkRunResult<TPayload = unknown, TArtifact = string> = BenchmarkRunResult<TPayload, TArtifact>;
391
- type index_BenchmarkScenario<TPayload = unknown> = BenchmarkScenario<TPayload>;
392
- type index_BenchmarkSliceSummary = BenchmarkSliceSummary;
393
- type index_BenchmarkSource = BenchmarkSource;
394
- type index_BenchmarkTaskKind = BenchmarkTaskKind;
395
- type index_BuildStandardRetrievalItemsOptions = BuildStandardRetrievalItemsOptions;
396
- type index_RetrievalIdAdapterOptions = RetrievalIdAdapterOptions;
397
- type index_StandardRetrievalArtifact = StandardRetrievalArtifact;
398
- type index_StandardRetrievalDocument = StandardRetrievalDocument;
399
- type index_StandardRetrievalEvaluationOptions = StandardRetrievalEvaluationOptions;
400
- type index_StandardRetrievalPayload = StandardRetrievalPayload;
401
- type index_StandardRetrievalQrel = StandardRetrievalQrel;
402
- type index_StandardRetrievalQuery = StandardRetrievalQuery;
403
- type index_StandardRetrievalResult = StandardRetrievalResult;
404
- declare const index_buildStandardRetrievalItems: typeof buildStandardRetrievalItems;
405
- declare const index_calibrateBenchmarkMetric: typeof calibrateBenchmarkMetric;
406
- declare const index_createRetrievalIdBenchmarkAdapter: typeof createRetrievalIdBenchmarkAdapter;
407
- declare const index_deterministicSplit: typeof deterministicSplit;
408
- declare const index_evaluateStandardRetrieval: typeof evaluateStandardRetrieval;
409
- declare const index_normalizeRetrievedDocumentIds: typeof normalizeRetrievedDocumentIds;
410
- declare const index_parseBeirCorpusJsonl: typeof parseBeirCorpusJsonl;
411
- declare const index_parseBeirQueriesJsonl: typeof parseBeirQueriesJsonl;
412
- declare const index_parseJsonlRows: typeof parseJsonlRows;
413
- declare const index_parseQrels: typeof parseQrels;
414
- declare const index_parseTsvRows: typeof parseTsvRows;
415
- declare const index_renderBenchmarkReportMarkdown: typeof renderBenchmarkReportMarkdown;
416
- declare const index_retrievalMetricsAtCutoff: typeof retrievalMetricsAtCutoff;
417
- declare const index_runBenchmarkAdapter: typeof runBenchmarkAdapter;
418
- declare const index_summarizeBenchmarkCampaign: typeof summarizeBenchmarkCampaign;
419
- declare namespace index {
420
- export { index_BENCHMARK_SPLIT_SEED as BENCHMARK_SPLIT_SEED, type index_BenchmarkAdapter as BenchmarkAdapter, type index_BenchmarkDatasetItem as BenchmarkDatasetItem, type index_BenchmarkDistribution as BenchmarkDistribution, type index_BenchmarkEvaluation as BenchmarkEvaluation, type index_BenchmarkFamily as BenchmarkFamily, type index_BenchmarkMetricCalibrationOptions as BenchmarkMetricCalibrationOptions, type index_BenchmarkMetricCalibrationResult as BenchmarkMetricCalibrationResult, type index_BenchmarkReport as BenchmarkReport, type index_BenchmarkResponder as BenchmarkResponder, type index_BenchmarkRunOptions as BenchmarkRunOptions, type index_BenchmarkRunResult as BenchmarkRunResult, type index_BenchmarkScenario as BenchmarkScenario, type index_BenchmarkSliceSummary as BenchmarkSliceSummary, type index_BenchmarkSource as BenchmarkSource, type index_BenchmarkTaskKind as BenchmarkTaskKind, type index_BuildStandardRetrievalItemsOptions as BuildStandardRetrievalItemsOptions, type index_RetrievalIdAdapterOptions as RetrievalIdAdapterOptions, type index_StandardRetrievalArtifact as StandardRetrievalArtifact, type index_StandardRetrievalDocument as StandardRetrievalDocument, type index_StandardRetrievalEvaluationOptions as StandardRetrievalEvaluationOptions, type index_StandardRetrievalPayload as StandardRetrievalPayload, type index_StandardRetrievalQrel as StandardRetrievalQrel, type index_StandardRetrievalQuery as StandardRetrievalQuery, type index_StandardRetrievalResult as StandardRetrievalResult, index_buildStandardRetrievalItems as buildStandardRetrievalItems, index_calibrateBenchmarkMetric as calibrateBenchmarkMetric, index_createRetrievalIdBenchmarkAdapter as createRetrievalIdBenchmarkAdapter, index_deterministicSplit as deterministicSplit, index_evaluateStandardRetrieval as evaluateStandardRetrieval, index_normalizeRetrievedDocumentIds as normalizeRetrievedDocumentIds, index_parseBeirCorpusJsonl as parseBeirCorpusJsonl, index_parseBeirQueriesJsonl as parseBeirQueriesJsonl, index_parseJsonlRows as parseJsonlRows, index_parseQrels as parseQrels, index_parseTsvRows as parseTsvRows, index_renderBenchmarkReportMarkdown as renderBenchmarkReportMarkdown, index_retrievalMetricsAtCutoff as retrievalMetricsAtCutoff, index$1 as routing, index_runBenchmarkAdapter as runBenchmarkAdapter, index_summarizeBenchmarkCampaign as summarizeBenchmarkCampaign };
421
- }
422
-
423
- export { createRetrievalIdBenchmarkAdapter as A, BENCHMARK_SPLIT_SEED as B, evaluateStandardRetrieval as C, normalizeRetrievedDocumentIds as D, parseBeirCorpusJsonl as E, parseBeirQueriesJsonl as F, parseJsonlRows as G, parseQrels as H, parseTsvRows as I, renderBenchmarkReportMarkdown as J, retrievalMetricsAtCutoff as K, index$1 as L, runBenchmarkAdapter as M, summarizeBenchmarkCampaign as N, type RetrievalIdAdapterOptions as R, type StandardRetrievalArtifact as S, type BenchmarkAdapter as a, type BenchmarkDatasetItem as b, type BenchmarkEvaluation as c, type BenchmarkFamily as d, type BenchmarkResponder as e, type BenchmarkScenario as f, type BenchmarkSource as g, type BenchmarkTaskKind as h, deterministicSplit as i, index as j, type BenchmarkDistribution as k, type BenchmarkMetricCalibrationOptions as l, type BenchmarkMetricCalibrationResult as m, type BenchmarkReport as n, type BenchmarkRunOptions as o, type BenchmarkRunResult as p, type BenchmarkSliceSummary as q, type BuildStandardRetrievalItemsOptions as r, type StandardRetrievalDocument as s, type StandardRetrievalEvaluationOptions as t, type StandardRetrievalPayload as u, type StandardRetrievalQrel as v, type StandardRetrievalQuery as w, type StandardRetrievalResult as x, buildStandardRetrievalItems as y, calibrateBenchmarkMetric as z };