@kimpton-ai/evalrouter 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,680 @@
1
+ export type BenchmarkDetail = {
2
+ slug: string;
3
+ versions: Array<BenchmarkRecord>;
4
+ };
5
+ export type BenchmarkNotice = {
6
+ adaptation_notice: string;
7
+ evidence_sha256: string;
8
+ licenses: Array<LicenseNotice>;
9
+ manifest_sha256: string;
10
+ profile_id: string;
11
+ };
12
+ export type BenchmarkQuoteAvailability = {
13
+ message?: string | null;
14
+ reason_code?: string | null;
15
+ status: "ready_for_quote" | "unavailable";
16
+ };
17
+ export type BenchmarkRecord = {
18
+ category?: string | null;
19
+ code_license?: string | null;
20
+ data_license?: string | null;
21
+ description?: string | null;
22
+ display_policy?: string | null;
23
+ fewshot_config?: {
24
+ [key: string]: JsonValue_Output;
25
+ } | null;
26
+ generation_config?: {
27
+ [key: string]: JsonValue_Output;
28
+ } | null;
29
+ id: string;
30
+ license_url?: string | null;
31
+ manifest_sha256: string;
32
+ metrics?: Array<string> | null;
33
+ name?: string | null;
34
+ provenance?: {
35
+ [key: string]: JsonValue_Output;
36
+ } | null;
37
+ quote_availability?: BenchmarkQuoteAvailability | null;
38
+ required_capabilities?: Array<string> | null;
39
+ runner: string;
40
+ sampling_algorithm?: string | null;
41
+ slug: string;
42
+ source_revision?: string | null;
43
+ source_url?: string | null;
44
+ status: string;
45
+ task_count?: number | null;
46
+ version: string;
47
+ };
48
+ export type BillingSnapshot = {
49
+ as_of: string;
50
+ cap_microusd: string;
51
+ pending: boolean;
52
+ reserved_microusd: string;
53
+ spend_microusd: string;
54
+ };
55
+ export type CatalogBenchmarksParams = {
56
+ q?: string;
57
+ runner?: "inspect" | "lm-eval" | "evalplus" | "livecodebench" | "spreadsheetbench" | null;
58
+ capability?: string | null;
59
+ status?: "candidate" | "active" | "quarantined" | "retired" | null;
60
+ cursor?: string | null;
61
+ limit?: number;
62
+ };
63
+ export type CatalogEnvironmentsParams = {
64
+ cursor?: string | null;
65
+ limit?: number;
66
+ };
67
+ export type CatalogEvalsParams = {
68
+ provider?: string | null;
69
+ kind?: "benchmark" | "environment" | "remote" | null;
70
+ protocol?: "batch" | "interactive" | "agentic" | null;
71
+ q?: string | null;
72
+ behavior?: string | null;
73
+ availability?: string | null;
74
+ publisher?: string | null;
75
+ include_private?: boolean;
76
+ cursor?: string | null;
77
+ limit?: number;
78
+ };
79
+ export type CatalogModelsParams = {
80
+ q?: string;
81
+ provider?: string | null;
82
+ capability?: string | null;
83
+ cursor?: string | null;
84
+ limit?: number;
85
+ };
86
+ export type ComparisonModel = {
87
+ kind: "comparison";
88
+ model_id: string;
89
+ models: Array<SelectedModel>;
90
+ };
91
+ export type ConnectedModel = {
92
+ connection_id: string;
93
+ kind: "connection";
94
+ };
95
+ export type ConnectionRecord = {
96
+ base_url: string;
97
+ capabilities: Array<string>;
98
+ check_result: {
99
+ [key: string]: JsonValue_Output;
100
+ };
101
+ checked_at: string | null;
102
+ config: {
103
+ [key: string]: JsonValue_Output;
104
+ };
105
+ created_at: string;
106
+ disabled_at: string | null;
107
+ id: string;
108
+ model_id: string;
109
+ name: string;
110
+ protocol: "openai_chat" | "openai_completions";
111
+ revision: number;
112
+ };
113
+ export type ConnectionsListParams = {
114
+ cursor?: string | null;
115
+ limit?: number;
116
+ };
117
+ export type CoverageCounts = {
118
+ attempted: number;
119
+ cancelled: number;
120
+ errors: number;
121
+ graded: number;
122
+ planned: number;
123
+ skipped_budget: number;
124
+ };
125
+ export type CoverageRecord = {
126
+ mode: "sample" | "full";
127
+ sample_count?: number | null;
128
+ seed?: number | null;
129
+ };
130
+ export type EnvironmentFamily = {
131
+ family: string;
132
+ versions: Array<EnvironmentRecord>;
133
+ };
134
+ export type EnvironmentLimits = {
135
+ cpu_millis?: number;
136
+ duration_seconds?: number;
137
+ max_observation_bytes?: number;
138
+ max_output_tokens?: number;
139
+ max_steps?: number;
140
+ memory_mib?: number;
141
+ network?: "blocked";
142
+ };
143
+ export type EnvironmentRecord = {
144
+ adapter: string;
145
+ commission_basis_points?: number;
146
+ created_at: string;
147
+ description: string;
148
+ id: string;
149
+ limitations: Array<string>;
150
+ limits: EnvironmentLimits;
151
+ manifest_sha256: string;
152
+ name: string;
153
+ package_sha256: string;
154
+ price_version: string;
155
+ quote_availability?: BenchmarkQuoteAvailability | null;
156
+ ref: string;
157
+ required_capabilities: Array<string>;
158
+ rights: EnvironmentRights;
159
+ status: string;
160
+ supplier_fee_microusd: string;
161
+ task_counts: {
162
+ [key: string]: number;
163
+ };
164
+ validation: {
165
+ [key: string]: JsonValue_Output;
166
+ };
167
+ visibility: string;
168
+ };
169
+ export type EnvironmentRights = {
170
+ attribution: string;
171
+ commercial_execution: boolean;
172
+ display: boolean;
173
+ export: boolean;
174
+ license: string;
175
+ source_url: string;
176
+ training: boolean;
177
+ underlying_licenses?: Array<string>;
178
+ };
179
+ export type EvalPricing = {
180
+ execution_fee_microusd?: string | null;
181
+ price_version: string;
182
+ task_fee_microusd?: string | null;
183
+ };
184
+ export type EvalRecord = {
185
+ availability?: string;
186
+ availability_reason?: string;
187
+ behaviors?: Array<string>;
188
+ category?: string;
189
+ description: string;
190
+ detail_href?: string | null;
191
+ display_name: string;
192
+ execution_mode: "router" | "provider";
193
+ family?: string;
194
+ kind: "benchmark" | "environment" | "remote";
195
+ name: string;
196
+ pricing: EvalPricing;
197
+ profile_id: string | null;
198
+ protocol: "batch" | "interactive" | "agentic";
199
+ provider: string;
200
+ publisher?: string;
201
+ ref: string;
202
+ required_capabilities: Array<string>;
203
+ runner: string;
204
+ source?: string;
205
+ status: string;
206
+ task_count: number | null;
207
+ version: string;
208
+ visibility?: "public" | "private";
209
+ };
210
+ export type EvalReference = {
211
+ eval: string;
212
+ provider?: string | null;
213
+ split?: "canonical" | "train" | "development" | "test" | null;
214
+ };
215
+ export type EvalsListParams = {
216
+ provider?: string | null;
217
+ kind?: "benchmark" | "environment" | "remote" | null;
218
+ protocol?: "batch" | "interactive" | "agentic" | null;
219
+ q?: string | null;
220
+ behavior?: string | null;
221
+ availability?: string | null;
222
+ publisher?: string | null;
223
+ include_private?: boolean;
224
+ cursor?: string | null;
225
+ limit?: number;
226
+ };
227
+ export type EventPage = {
228
+ data: Array<RunEvent>;
229
+ has_more: boolean;
230
+ next_sequence: number;
231
+ };
232
+ export type ExecutionOptions = {
233
+ chat_enabled?: boolean;
234
+ max_models_per_run: number;
235
+ };
236
+ export type FullCoverage = {
237
+ mode: "full";
238
+ };
239
+ export type Interval = {
240
+ approximate: boolean;
241
+ cluster_count?: number | null;
242
+ confidence: number;
243
+ eligible_count: number;
244
+ gains?: number | null;
245
+ losses?: number | null;
246
+ lower: number;
247
+ method: string;
248
+ resamples?: number | null;
249
+ sample_count: number;
250
+ seed?: number | null;
251
+ strata_count: number;
252
+ target: string;
253
+ unrepresented_strata: number;
254
+ upper: number;
255
+ weights: string;
256
+ };
257
+ export type JsonValue = JsonValue_Output;
258
+ export type JsonValue_Output = string | number | boolean | Array<JsonValue_Output> | {
259
+ [key: string]: JsonValue_Output;
260
+ } | null;
261
+ export type JudgeSelection = {
262
+ model: SelectedModel;
263
+ protocol: {
264
+ [key: string]: JsonValue_Output;
265
+ };
266
+ };
267
+ export type LicenseNotice = {
268
+ attribution: string;
269
+ document_sha256: string;
270
+ identifier: string;
271
+ scope: "code" | "data" | "tests" | "assets";
272
+ source_revision: string;
273
+ source_url: string;
274
+ text: string;
275
+ };
276
+ export type ManagedModel = {
277
+ kind: "managed";
278
+ route_id: string;
279
+ };
280
+ export type ManagedRoute = {
281
+ capabilities: Array<string>;
282
+ display_name: string;
283
+ generation?: {
284
+ [key: string]: JsonValue_Output;
285
+ } | null;
286
+ hosting_provider?: string | null;
287
+ id: string;
288
+ limits?: {
289
+ [key: string]: JsonValue_Output;
290
+ } | null;
291
+ model_id: string;
292
+ price_version: string;
293
+ pricing: {
294
+ [key: string]: JsonValue_Output;
295
+ };
296
+ provider: string;
297
+ revision: number;
298
+ status: string;
299
+ };
300
+ export type MetricRecord = {
301
+ direction: "higher_is_better" | "lower_is_better" | null;
302
+ key: string;
303
+ name: string;
304
+ sample_count: number | null;
305
+ source: string;
306
+ standard_error: number | null;
307
+ task: string;
308
+ uncertainty?: Uncertainty | null;
309
+ value: number;
310
+ };
311
+ export type NewConnection = {
312
+ api_key: string;
313
+ base_url: string;
314
+ context_window?: number;
315
+ max_output_tokens?: number;
316
+ model_id: string;
317
+ name: string;
318
+ protocol?: "openai_chat" | "openai_completions";
319
+ };
320
+ export type NewEvaluation = {
321
+ budget_usd?: number | string | null;
322
+ coverage?: SampleCoverage | FullCoverage | null;
323
+ eval: string;
324
+ max_charge_microusd?: string | null;
325
+ metadata?: {
326
+ [key: string]: string;
327
+ };
328
+ model: string | ManagedModel | ConnectedModel;
329
+ name?: string;
330
+ provider?: string | null;
331
+ split?: string | null;
332
+ };
333
+ export type NewQuote = {
334
+ selection: Profiles | EvalReference;
335
+ model: ManagedModel | ConnectedModel;
336
+ coverage?: SampleCoverage | FullCoverage | null;
337
+ max_charge_microusd: string;
338
+ };
339
+ export type NewRun = {
340
+ metadata?: {
341
+ [key: string]: string;
342
+ };
343
+ name?: string;
344
+ quote_id: string;
345
+ };
346
+ export type OKResponse = {
347
+ ok: boolean;
348
+ };
349
+ export type Page_BenchmarkRecord_ = {
350
+ data: Array<BenchmarkRecord>;
351
+ next_cursor: string | null;
352
+ };
353
+ export type Page_ConnectionRecord_ = {
354
+ data: Array<ConnectionRecord>;
355
+ next_cursor: string | null;
356
+ };
357
+ export type Page_EnvironmentRecord_ = {
358
+ data: Array<EnvironmentRecord>;
359
+ next_cursor: string | null;
360
+ };
361
+ export type Page_EvalRecord_ = {
362
+ data: Array<EvalRecord>;
363
+ next_cursor: string | null;
364
+ };
365
+ export type Page_ManagedRoute_ = {
366
+ data: Array<ManagedRoute>;
367
+ next_cursor: string | null;
368
+ };
369
+ export type Page_RunRecord_ = {
370
+ data: Array<RunRecord>;
371
+ next_cursor: string | null;
372
+ };
373
+ export type ProfileCost = {
374
+ charged_microusd: string;
375
+ model_index?: number | null;
376
+ pending_operations: number;
377
+ profile_id: string;
378
+ total_cost_microusd: string | null;
379
+ };
380
+ export type ProfileResult = {
381
+ cancelled: number;
382
+ data_license: string | null;
383
+ display_policy?: string | null;
384
+ eligible: number | null;
385
+ error_code: string | null;
386
+ errors: number;
387
+ graded: number;
388
+ headline_key: string | null;
389
+ license_url: string | null;
390
+ metrics: Array<MetricRecord>;
391
+ model_index?: number | null;
392
+ name: string;
393
+ native: {
394
+ [key: string]: JsonValue_Output;
395
+ } | null;
396
+ planned: number;
397
+ profile_id: string;
398
+ runner: string;
399
+ selection: ProfileSelection;
400
+ skipped_budget: number;
401
+ source_revision: string | null;
402
+ source_url: string | null;
403
+ status: string;
404
+ uncertainty?: {
405
+ [key: string]: Uncertainty;
406
+ } | null;
407
+ uncertainty_note: string;
408
+ unit_id: string;
409
+ };
410
+ export type ProfileSelection = {
411
+ admission?: {
412
+ [key: string]: JsonValue_Output;
413
+ } | null;
414
+ attempts_per_task?: number;
415
+ dataset_assets_sha256: string;
416
+ environment?: {
417
+ [key: string]: JsonValue_Output;
418
+ } | null;
419
+ id: string;
420
+ judge?: JudgeSelection | null;
421
+ manifest_sha256: string;
422
+ population_count: number;
423
+ private_test?: boolean | null;
424
+ remote?: {
425
+ [key: string]: JsonValue_Output;
426
+ } | null;
427
+ sampling_algorithm: string;
428
+ sandbox?: SandboxPolicyView | null;
429
+ selected_count: number;
430
+ task_ids?: Array<string> | null;
431
+ task_ids_hash: string;
432
+ };
433
+ export type Profiles = {
434
+ profile_ids: Array<string>;
435
+ };
436
+ export type QuotePlan = {
437
+ agent?: {
438
+ [key: string]: JsonValue_Output;
439
+ } | null;
440
+ agents?: Array<{
441
+ [key: string]: JsonValue_Output;
442
+ }> | null;
443
+ coverage: CoverageRecord;
444
+ model: SelectedModel | ComparisonModel;
445
+ pricing: {
446
+ [key: string]: JsonValue_Output;
447
+ };
448
+ profiles: Array<ProfileSelection>;
449
+ recommendation?: {
450
+ [key: string]: JsonValue_Output;
451
+ } | null;
452
+ route?: {
453
+ [key: string]: JsonValue_Output;
454
+ } | null;
455
+ schema_version: 1 | 2 | 3 | 4;
456
+ selection_source: {
457
+ [key: string]: JsonValue_Output;
458
+ };
459
+ total_tasks: number;
460
+ warnings: Array<string>;
461
+ };
462
+ export type QuoteRecord = {
463
+ created_at: string;
464
+ estimated_charge_microusd: string;
465
+ expires_at: string;
466
+ id: string;
467
+ max_charge_microusd: string;
468
+ plan: QuotePlan;
469
+ price_version: string;
470
+ retention: RetentionPolicy;
471
+ };
472
+ export type ResultCosts = {
473
+ as_of: string;
474
+ basis: "managed_total" | "platform_fees_only" | "mixed";
475
+ currency: "USD";
476
+ note: string;
477
+ profiles: Array<ProfileCost>;
478
+ };
479
+ export type ResultRecord = {
480
+ benchmark_notices: Array<BenchmarkNotice>;
481
+ checksum: string;
482
+ costs: ResultCosts;
483
+ coverage: CoverageRecord | null;
484
+ envelope?: {
485
+ [key: string]: JsonValue_Output;
486
+ } | null;
487
+ finalized_at: string;
488
+ id: string;
489
+ manifest_hash: string;
490
+ model: SelectedModel | ComparisonModel;
491
+ profiles: Array<ProfileResult>;
492
+ retention: RunRetention;
493
+ run_id: string;
494
+ summary: ResultSummary;
495
+ version: number;
496
+ };
497
+ export type ResultSummary = {
498
+ billing_snapshot: BillingSnapshot;
499
+ coverage: CoverageCounts;
500
+ finished_at: string;
501
+ manifest_hash: string;
502
+ name: string;
503
+ result_version: number;
504
+ run_id: string;
505
+ schema_version: 1;
506
+ status: "completed" | "partial" | "failed" | "cancelled";
507
+ terminal_reason: string | null;
508
+ };
509
+ export type RetentionPolicy = {
510
+ raw_days: number;
511
+ starts_at: "run_created_at";
512
+ summary_days: number;
513
+ };
514
+ export type RunEvent = {
515
+ created_at: string;
516
+ data: {
517
+ [key: string]: JsonValue_Output;
518
+ };
519
+ sequence: number;
520
+ type: string;
521
+ };
522
+ export type RunReceipt = {
523
+ billing_state: "pending" | "settled" | "reconciliation_required";
524
+ cap_microusd: string;
525
+ created_at: string;
526
+ currency: "USD";
527
+ report_expired: boolean;
528
+ reserved_microusd: string;
529
+ run_id: string;
530
+ spend_microusd: string;
531
+ summary_expires_at: string;
532
+ };
533
+ export type RunRecord = {
534
+ agent?: {
535
+ [key: string]: JsonValue_Output;
536
+ } | null;
537
+ agents?: Array<{
538
+ [key: string]: JsonValue_Output;
539
+ }> | null;
540
+ billing_state: "pending" | "settled" | "reconciliation_required";
541
+ cancel_requested_at: string | null;
542
+ cap_microusd: string;
543
+ completed_count: number;
544
+ coverage: CoverageRecord | null;
545
+ created_at: string;
546
+ error_count: number;
547
+ finished_at: string | null;
548
+ id: string;
549
+ manifest_hash: string;
550
+ metadata: {
551
+ [key: string]: string;
552
+ };
553
+ model: SelectedModel | ComparisonModel;
554
+ name: string;
555
+ planned_count: number;
556
+ profiles: Array<ProfileSelection>;
557
+ reserved_microusd: string;
558
+ retention: RunRetention;
559
+ retry_at?: string | null;
560
+ route?: {
561
+ [key: string]: JsonValue_Output;
562
+ } | null;
563
+ spend_microusd: string;
564
+ started_at: string | null;
565
+ status: "queued" | "running" | "cancelling" | "finalizing" | "completed" | "partial" | "failed" | "cancelled";
566
+ terminal_reason: string | null;
567
+ waiting_reason?: string | null;
568
+ };
569
+ export type RunRetention = {
570
+ raw_evidence_expired: boolean;
571
+ raw_expires_at: string;
572
+ summary_expires_at: string;
573
+ };
574
+ export type RunsEventsParams = {
575
+ after?: number;
576
+ limit?: number;
577
+ tail?: boolean;
578
+ };
579
+ export type RunsExecutionsParams = {
580
+ sample_id?: string | null;
581
+ cursor?: string | null;
582
+ limit?: number;
583
+ };
584
+ export type RunsExportParams = {
585
+ format?: "json" | "csv" | "html";
586
+ version?: number | null;
587
+ };
588
+ export type RunsListParams = {
589
+ cursor?: string | null;
590
+ limit?: number;
591
+ };
592
+ export type RunsResultsParams = {
593
+ version?: number | null;
594
+ };
595
+ export type SampleCoverage = {
596
+ mode?: "sample";
597
+ sample_count?: number;
598
+ seed?: number;
599
+ };
600
+ export type SandboxExecutionPage = {
601
+ data: Array<SandboxExecutionRecord>;
602
+ next_cursor: string | null;
603
+ summary: SandboxExecutionSummary;
604
+ };
605
+ export type SandboxExecutionRecord = {
606
+ attempt_no: number;
607
+ cleanup_status: "not_requested" | "pending" | "stopped";
608
+ cpu_millis: number;
609
+ created_at: string;
610
+ error_code: string | null;
611
+ execution_started_at: string | null;
612
+ id: string;
613
+ memory_mib: number;
614
+ model_index: number;
615
+ profile_id: string;
616
+ sample_id: string;
617
+ status: "intent" | "creating" | "ready" | "executing" | "finished" | "stopping" | "stopped";
618
+ stopped_at: string | null;
619
+ task_id: string;
620
+ timeout_seconds: number;
621
+ verifier_result_recorded: boolean;
622
+ };
623
+ export type SandboxExecutionSummary = {
624
+ active_count: number;
625
+ allocated_count: number;
626
+ cleanup_pending_count: number;
627
+ stopped_count: number;
628
+ };
629
+ export type SandboxPolicyView = {
630
+ backend?: string;
631
+ cpu_millis?: number;
632
+ execution_class?: string;
633
+ execution_fee_microusd: string;
634
+ image_id: string;
635
+ image_sha256: string;
636
+ max_infrastructure_attempts?: number;
637
+ memory_mib?: number;
638
+ network?: string;
639
+ price_version: string;
640
+ schema_version?: 1;
641
+ timeout_seconds?: number;
642
+ };
643
+ export type SelectedModel = {
644
+ artifact_id?: string | null;
645
+ artifact_sha256?: string | null;
646
+ capabilities: Array<string>;
647
+ connection_id?: string | null;
648
+ generation?: {
649
+ [key: string]: JsonValue_Output;
650
+ } | null;
651
+ kind: string;
652
+ likelihood?: {
653
+ [key: string]: JsonValue_Output;
654
+ } | null;
655
+ limits: {
656
+ [key: string]: JsonValue_Output;
657
+ };
658
+ model_id: string;
659
+ price_version?: string | null;
660
+ pricing?: {
661
+ [key: string]: JsonValue_Output;
662
+ } | null;
663
+ protocol: string;
664
+ provider_charges?: string | null;
665
+ revision: number;
666
+ route_id?: string | null;
667
+ routing?: {
668
+ [key: string]: JsonValue_Output;
669
+ } | null;
670
+ };
671
+ export type Uncertainty = {
672
+ interval: Interval | null;
673
+ reason: string | null;
674
+ };
675
+ export type UpdateConnection = {
676
+ api_key?: string | null;
677
+ context_window?: number | null;
678
+ max_output_tokens?: number | null;
679
+ name?: string | null;
680
+ };
package/dist/types.js ADDED
@@ -0,0 +1,2 @@
1
+ // Generated by scripts/generate_clients.py; OpenAPI SHA256 3cea8bc29adf3dd7ab60f48b418a77478e3e4502357e1e23de4a2f7e97088a3f.
2
+ export {};
package/package.json ADDED
@@ -0,0 +1,29 @@
1
+ {
2
+ "name": "@kimpton-ai/evalrouter",
3
+ "version": "0.1.0",
4
+ "license": "SEE LICENSE IN LICENSE",
5
+ "repository": {
6
+ "type": "git",
7
+ "url": "git+https://github.com/kimpton-ai/kimpton-evalrouter.git"
8
+ },
9
+ "type": "module",
10
+ "description": "Typed Kimpton evaluation client for trusted Node.js applications",
11
+ "engines": {
12
+ "node": ">=22"
13
+ },
14
+ "exports": {
15
+ ".": {
16
+ "types": "./dist/index.d.ts",
17
+ "import": "./dist/index.js"
18
+ },
19
+ "./types": {
20
+ "types": "./dist/types.d.ts",
21
+ "import": "./dist/types.js"
22
+ }
23
+ },
24
+ "files": [
25
+ "dist",
26
+ "README.md",
27
+ "LICENSE"
28
+ ]
29
+ }