evalrouter 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalrouter-0.3.0.dist-info/METADATA +267 -0
- evalrouter-0.3.0.dist-info/RECORD +33 -0
- evalrouter-0.3.0.dist-info/WHEEL +4 -0
- evalrouter-0.3.0.dist-info/entry_points.txt +2 -0
- evalrouter-0.3.0.dist-info/licenses/LICENSE +29 -0
- evalrouter_contracts/__init__.py +11 -0
- evalrouter_contracts/bundle.py +46 -0
- evalrouter_contracts/components.py +181 -0
- evalrouter_contracts/models.py +288 -0
- evalrouter_contracts/protocol.py +221 -0
- evalrouter_contracts/py.typed +0 -0
- evalrouter_contracts/templates.py +136 -0
- evalrouter_contracts/validation.py +329 -0
- kimpton_evalrouter/__init__.py +112 -0
- kimpton_evalrouter/__main__.py +3 -0
- kimpton_evalrouter/_account.py +1379 -0
- kimpton_evalrouter/_boundary.py +191 -0
- kimpton_evalrouter/_guided.py +1889 -0
- kimpton_evalrouter/_helpers.py +101 -0
- kimpton_evalrouter/_listing.py +1107 -0
- kimpton_evalrouter/_resources.py +537 -0
- kimpton_evalrouter/_transport.py +436 -0
- kimpton_evalrouter/cli.py +1676 -0
- kimpton_evalrouter/py.typed +0 -0
- kimpton_evalrouter/render/__init__.py +99 -0
- kimpton_evalrouter/render/_console.py +91 -0
- kimpton_evalrouter/render/_theme.py +80 -0
- kimpton_evalrouter/render/_widgets.py +132 -0
- kimpton_evalrouter/render/live.py +514 -0
- kimpton_evalrouter/render/picker.py +230 -0
- kimpton_evalrouter/render/screens.py +842 -0
- kimpton_evalrouter/render/tables.py +427 -0
- kimpton_evalrouter/types.py +1171 -0
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: evalrouter
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Typed Kimpton evaluation client and command-line interface
|
|
5
|
+
License-Expression: LicenseRef-Kimpton-EvalRouter-SDK
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Requires-Python: <3.14,>=3.12
|
|
8
|
+
Requires-Dist: httpx==0.28.1
|
|
9
|
+
Requires-Dist: rich==15.0.0
|
|
10
|
+
Provides-Extra: benchmark
|
|
11
|
+
Requires-Dist: pydantic==2.13.5; extra == 'benchmark'
|
|
12
|
+
Provides-Extra: keyring
|
|
13
|
+
Requires-Dist: keyring==25.7.0; extra == 'keyring'
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
|
|
16
|
+
# EvalRouter CLI
|
|
17
|
+
|
|
18
|
+
Evaluate models or qualified repository agents from your terminal: discover
|
|
19
|
+
supported targets, review a spending cap, run an evaluation, and export results. Requires Python 3.12
|
|
20
|
+
or 3.13. Distributed under the proprietary [Kimpton EvalRouter SDK License](LICENSE).
|
|
21
|
+
|
|
22
|
+
## Install and create your account
|
|
23
|
+
|
|
24
|
+
We recommend [uv](https://docs.astral.sh/uv/getting-started/installation/) to install the CLI in its own Python environment:
|
|
25
|
+
|
|
26
|
+
```sh
|
|
27
|
+
uv tool install --python 3.12 evalrouter
|
|
28
|
+
evalrouter --help
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
With pip, use `pip install evalrouter`. uv supplies Python 3.12 if needed. If
|
|
32
|
+
your shell cannot find `evalrouter`, run `uv tool update-shell` and restart your
|
|
33
|
+
terminal. Upgrade an existing installation with `uv tool upgrade evalrouter`.
|
|
34
|
+
|
|
35
|
+
This package was previously published as `kimpton-evalrouter-sdk`. The command,
|
|
36
|
+
the `kimpton_evalrouter` import and every API are unchanged. To switch, run
|
|
37
|
+
`uv tool uninstall kimpton-evalrouter-sdk` (or
|
|
38
|
+
`pip uninstall -y kimpton-evalrouter-sdk evalrouter`) and install `evalrouter`.
|
|
39
|
+
|
|
40
|
+
The package supplies the `evalrouter` command; no repository checkout is needed.
|
|
41
|
+
[Create an EvalRouter account](https://evalrouter.ai/register), verify your email,
|
|
42
|
+
and create a workspace key in [API key settings](https://evalrouter.ai/settings/api-keys).
|
|
43
|
+
Package installation does not grant account access or evaluation credit.
|
|
44
|
+
|
|
45
|
+
Set `EVALROUTER_API_KEY` and `EVALROUTER_WORKSPACE_ID` through your secret manager
|
|
46
|
+
or private environment. Use `https://api.evalrouter.ai` for `EVALROUTER_BASE_URL`,
|
|
47
|
+
without `/v1`. Never put credentials in arguments, URLs, request files or logs.
|
|
48
|
+
The CLI uses an existing workspace key; browser sign-up is a separate step.
|
|
49
|
+
|
|
50
|
+
## Run an evaluation in one command
|
|
51
|
+
|
|
52
|
+
```sh
|
|
53
|
+
evalrouter run gpqa-diamond --model gpt-4o-mini
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
`run` matches the benchmark and model by slug, name or ID, suggesting the
|
|
57
|
+
closest names after a typo. It creates a free quote, shows the coverage,
|
|
58
|
+
conservative estimate, spending cap and concurrency, and asks `Run it? [Y/n]`.
|
|
59
|
+
Only a yes starts paid work. It then follows progress and prints the scores,
|
|
60
|
+
the change from your previous completed run of the same benchmark and model,
|
|
61
|
+
and the report link. In a terminal, missing choices are asked for; run
|
|
62
|
+
`evalrouter run` alone to pick everything. `--sample N` or `--full` sets
|
|
63
|
+
coverage, `--max-cost USD` the cap and `--dry-run` stops after the quote.
|
|
64
|
+
|
|
65
|
+
Without a terminal (CI, agents, `--json` or `--no-input`) `run` never prompts
|
|
66
|
+
and refuses unless both `--yes` and `--max-cost` are given:
|
|
67
|
+
|
|
68
|
+
```sh
|
|
69
|
+
evalrouter run gpqa-diamond --model gpt-4o-mini --max-cost 5 --yes --json
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
The idempotency key of each accepted quote is saved locally before the run is
|
|
73
|
+
submitted (`~/.local/state/evalrouter`, or `EVALROUTER_STATE_DIR`; identifiers
|
|
74
|
+
only, never credentials). Repeating the same command after a dropped connection
|
|
75
|
+
resumes that run instead of starting another; `--new` starts a separate one.
|
|
76
|
+
Afterwards, `results`, `export --format html,json`, `wait` and `open` default to
|
|
77
|
+
your last run and accept a unique run ID prefix. Watching a run survives brief
|
|
78
|
+
API outages such as a 503 during a deployment.
|
|
79
|
+
|
|
80
|
+
## Lists
|
|
81
|
+
|
|
82
|
+
`catalog`, `catalog --models`, `status`, `connections list` and `workspace list`
|
|
83
|
+
print a compact table sized to your terminal with one suggested next command;
|
|
84
|
+
`catalog` hides benchmarks in review or unavailable unless you pass `--all`, and
|
|
85
|
+
`catalog -i` browses them interactively. Every list accepts `--json`,
|
|
86
|
+
`--jq '.data[].id'` (a small jq subset), `-w/--wide`, `--columns a,b`,
|
|
87
|
+
`--sort COL` (`--sort=-COL` descending), `--filter KEY=VALUE` and `--limit N`.
|
|
88
|
+
Piped output stays JSON. Colour follows `NO_COLOR`, `CI` and `TERM=dumb`.
|
|
89
|
+
|
|
90
|
+
`run` shows the quote (expected cost from your previous run when known, worst
|
|
91
|
+
case, and the smallest workable cap) before asking for the cap, refuses a cap
|
|
92
|
+
that cannot fit one task, and asks before repeating a run that just finished
|
|
93
|
+
(scripts pass `--new`). `--all-metrics` lists every metric and comparison.
|
|
94
|
+
|
|
95
|
+
## Discover, quote, then run
|
|
96
|
+
|
|
97
|
+
```sh
|
|
98
|
+
evalrouter catalog --status active --json
|
|
99
|
+
evalrouter catalog --models --json
|
|
100
|
+
evalrouter catalog --slug BENCHMARK_SLUG --json
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Choose a profile whose `quote_availability.status` is `ready_for_quote` and a
|
|
104
|
+
compatible managed model route from those responses. An active catalog entry
|
|
105
|
+
alone does not mean it is ready to run.
|
|
106
|
+
Save this as `quote.json`, replacing both uppercase identifiers:
|
|
107
|
+
|
|
108
|
+
```json
|
|
109
|
+
{
|
|
110
|
+
"model": {"kind": "managed", "route_id": "MODEL_ROUTE_ID"},
|
|
111
|
+
"selection": {"profile_ids": ["BENCHMARK_PROFILE_ID"]},
|
|
112
|
+
"coverage": {"mode": "sample", "sample_count": 3, "seed": 42},
|
|
113
|
+
"max_charge_microusd": "1000000"
|
|
114
|
+
}
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
```sh
|
|
118
|
+
evalrouter quote --config quote.json --json
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
Review compatibility, coverage, cost components, warnings and expiry. A quote
|
|
122
|
+
starts no paid work. The example's $1 platform cap is not a price guarantee or
|
|
123
|
+
promise that a particular evaluation fits. Paid runs require available credit.
|
|
124
|
+
|
|
125
|
+
Persist the returned quote ID and an operation key before submitting:
|
|
126
|
+
|
|
127
|
+
```sh
|
|
128
|
+
evalrouter run --quote REVIEWED_QUOTE_ID --idempotency-key SAVED_OPERATION_KEY --json
|
|
129
|
+
evalrouter wait RUN_ID --wait-timeout 3600 --json
|
|
130
|
+
evalrouter results RUN_ID --json
|
|
131
|
+
evalrouter export RUN_ID --format json --output result.json
|
|
132
|
+
evalrouter export RUN_ID --format csv --output result.csv
|
|
133
|
+
evalrouter export RUN_ID --format html --output result.html
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
Replace `RUN_ID` with the returned ID. A lost submission response is recovered
|
|
137
|
+
with the same quote and operation key; a new key may start separate work.
|
|
138
|
+
Inspect terminal status, coverage, errors and billing alongside scores. A sample
|
|
139
|
+
is not a full-benchmark score. Exports refuse overwrite by default; use the
|
|
140
|
+
result's integer `--version` for repeatable reports.
|
|
141
|
+
|
|
142
|
+
## Supported commands
|
|
143
|
+
|
|
144
|
+
`catalog` (`--models`, `--slug` or `--environment`), `connections list/create/check/update/disable`,
|
|
145
|
+
`quote`, `run`, `status`, `wait`, `cancel`, `results`, `export` and `open`.
|
|
146
|
+
Repository qualification uses `agent-builds options/preview/create/list/status/cancel`.
|
|
147
|
+
Authoring your own benchmark package uses `benchmark init/validate/bundle`.
|
|
148
|
+
Run each command with `--help` for exact flags. A connected model's credential
|
|
149
|
+
is never a command argument: `connections create` asks with hidden input in a
|
|
150
|
+
terminal, and scripts use `--key-stdin` or `--key-env NAME`
|
|
151
|
+
(`connections update --rotate-key` replaces it). Connection checks can send
|
|
152
|
+
a small model request, and your model provider may bill separately.
|
|
153
|
+
|
|
154
|
+
For `wait` and `run --wait`, `--progress auto` shows elapsed time and status in
|
|
155
|
+
an interactive terminal, with a percentage when validated processed and planned
|
|
156
|
+
counts are available. Use `--progress plain` for structured progress or
|
|
157
|
+
`--progress off` to suppress it. Redirected stderr and `--json` retain structured
|
|
158
|
+
progress. A progress percentage counts processed samples; it is not a score.
|
|
159
|
+
|
|
160
|
+
Use `--config -` for JSON on stdin. `--json` writes one result on stdout and
|
|
161
|
+
progress on stderr. Exit codes: 0 success; 1 API/transport or failed run;
|
|
162
|
+
2 invalid input or rejected request (including a refused unattended `run`);
|
|
163
|
+
4 partial/cancelled waited run; 5 confirmation declined, nothing started. Interrupting
|
|
164
|
+
a local wait does not cancel server work. Use `evalrouter cancel RUN_ID`
|
|
165
|
+
explicitly when cancellation is intended.
|
|
166
|
+
|
|
167
|
+
See the [CLI documentation](https://evalrouter.ai/developers/docs) for account
|
|
168
|
+
setup, model selection, spending semantics, recovery and versioned exports.
|
|
169
|
+
The CLI uses the [model API](https://evalrouter.ai/developers/docs/model-api),
|
|
170
|
+
which is also documented for direct HTTP integrations.
|
|
171
|
+
|
|
172
|
+
## Author your own benchmark
|
|
173
|
+
|
|
174
|
+
`benchmark init`, `benchmark validate` and `benchmark bundle` run entirely on
|
|
175
|
+
your machine: no account, no network request, and nothing from your package is
|
|
176
|
+
imported or executed. The normative validator ships inside `evalrouter`, and
|
|
177
|
+
its one dependency (pydantic) is an optional extra so the base installation
|
|
178
|
+
stays thin (httpx and rich only). With pip, use `pip install 'evalrouter[benchmark]'`:
|
|
179
|
+
|
|
180
|
+
```sh
|
|
181
|
+
uv tool install --python 3.12 "evalrouter[benchmark]"
|
|
182
|
+
evalrouter benchmark init ./my-benchmark --namespace example --name my-benchmark
|
|
183
|
+
evalrouter benchmark validate ./my-benchmark
|
|
184
|
+
evalrouter benchmark bundle ./my-benchmark --output my-benchmark.zip
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
Without the extra these three commands stop before reading your package and
|
|
188
|
+
report `benchmark_extra_required` with the exact install string; `--help` and
|
|
189
|
+
every network command keep working. `init` writes an explicitly synthetic draft
|
|
190
|
+
and claims no license: replace the tasks, rights, references and limits before
|
|
191
|
+
submitting. Local validation is not platform admission — EvalRouter revalidates
|
|
192
|
+
every submission on its own servers.
|
|
193
|
+
|
|
194
|
+
## Repository agents
|
|
195
|
+
|
|
196
|
+
Repository commands require version 0.2.0 or later and a service that has enabled
|
|
197
|
+
repository qualification for the workspace. Installing the package does not
|
|
198
|
+
enable that service. Check its current availability and supported targets first:
|
|
199
|
+
|
|
200
|
+
```sh
|
|
201
|
+
evalrouter agent-builds options --json
|
|
202
|
+
evalrouter catalog --environment ENVIRONMENT_FAMILY_PATH --json
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
The catalog lookup takes the family path before `@`; qualification inputs still
|
|
206
|
+
require the exact versioned reference returned by options. Stop if qualification
|
|
207
|
+
is disabled or the desired exact target/model is absent.
|
|
208
|
+
The supported runtime is locked Node/npm with an `evalrouter-agent.json` manifest;
|
|
209
|
+
custom images and Python agent runtimes are not supported. Save `build.json`:
|
|
210
|
+
|
|
211
|
+
```json
|
|
212
|
+
{
|
|
213
|
+
"repository_url": "YOUR_PUBLIC_GITHUB_REPOSITORY_URL",
|
|
214
|
+
"ref": "main",
|
|
215
|
+
"manifest_path": "evalrouter-agent.json",
|
|
216
|
+
"qualification": {
|
|
217
|
+
"environment_ref": "EXACT_ENVIRONMENT_FROM_OPTIONS",
|
|
218
|
+
"model": {"kind": "managed", "route_id": "MODEL_FROM_OPTIONS"}
|
|
219
|
+
},
|
|
220
|
+
"max_cost_microusd": "1000000"
|
|
221
|
+
}
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
For a private repository, connect GitHub in the web application and explicitly
|
|
225
|
+
select that repository for this workspace. Add
|
|
226
|
+
`source: {"github_connection_id": "CONNECTION_ID", "github_repository_id": 123}`
|
|
227
|
+
using the two actual IDs from that connection. These IDs select a grant; they
|
|
228
|
+
are not credentials. The CLI does not accept GitHub tokens or authorize access.
|
|
229
|
+
|
|
230
|
+
```sh
|
|
231
|
+
evalrouter agent-builds preview --config build.json --json
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
Preview resolves an immutable commit and returns the qualification scope, costs,
|
|
235
|
+
cap and expiry. It creates no job or credit hold and starts no sandbox/model
|
|
236
|
+
work. Review it, then save the quote ID and a durable operation key before:
|
|
237
|
+
|
|
238
|
+
```sh
|
|
239
|
+
evalrouter agent-builds create --quote REVIEWED_BUILD_QUOTE --idempotency-key SAVED_BUILD_KEY --json
|
|
240
|
+
evalrouter agent-builds status BUILD_JOB_ID --json
|
|
241
|
+
evalrouter agent-builds list --quote REVIEWED_BUILD_QUOTE --json
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
Creation reserves the quoted cap and may incur its stated costs. Recover an
|
|
245
|
+
unknown submission with the same quote and key or the quote-filtered list;
|
|
246
|
+
a new key may create new work. Cancel explicitly with
|
|
247
|
+
`evalrouter agent-builds cancel BUILD_JOB_ID`. A `cancelling` or `reconciling`
|
|
248
|
+
status remains unresolved. Wait for terminal cleanup and zero held funds.
|
|
249
|
+
|
|
250
|
+
Ready requires `status: ready`, `cleanup_confirmed: true`, zero held funds and
|
|
251
|
+
an `agent.ref`. It establishes only the returned scope, not benchmark quality.
|
|
252
|
+
Use that record's exact `qualification.environment_ref`, `qualification.split`
|
|
253
|
+
and `qualification.evaluation_coverage` in a separate evaluation quote:
|
|
254
|
+
|
|
255
|
+
```json
|
|
256
|
+
{
|
|
257
|
+
"agent": "READY_AGENT_REF",
|
|
258
|
+
"selection": {"environment": "RETURNED_ENVIRONMENT_REF", "split": "RETURNED_SPLIT"},
|
|
259
|
+
"coverage": "REPLACE_WITH_RETURNED_EVALUATION_COVERAGE_OBJECT",
|
|
260
|
+
"max_charge_microusd": "1000000"
|
|
261
|
+
}
|
|
262
|
+
```
|
|
263
|
+
|
|
264
|
+
Replace the coverage placeholder with the entire returned object; do not guess
|
|
265
|
+
tasks, counts or seeds. Use the existing `quote`, `run`, `wait`, `results` and
|
|
266
|
+
`export` commands above. Qualification and evaluation have separate caps and
|
|
267
|
+
charges. A qualification pass does not submit an evaluation automatically.
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
evalrouter_contracts/__init__.py,sha256=QEe5ymjfZc8zmGw1hV3kHeciL0QRTYUSQz03eDJ7Ngk,568
|
|
2
|
+
evalrouter_contracts/bundle.py,sha256=3yW73EtwEY_dIa8sIGIfPGXkxbJvcEZzk7pocESRUdU,2109
|
|
3
|
+
evalrouter_contracts/components.py,sha256=FyqOf_nOgczd4H4UxFBffaax8tXkZX4M_uHD6-A8vxU,6495
|
|
4
|
+
evalrouter_contracts/models.py,sha256=PLNSvlVG_-mSSUDXUqA6QeLeEEeExCukDpNu844tGHs,10673
|
|
5
|
+
evalrouter_contracts/protocol.py,sha256=iBDMqlK7JkH0xPVr7mpKIoGDyUfoxVU9x9OboxeOcb0,8941
|
|
6
|
+
evalrouter_contracts/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
7
|
+
evalrouter_contracts/templates.py,sha256=Jn2aCPuy_LY3BV5PmRipqifse4KuFr8oQlmak9uszi0,5390
|
|
8
|
+
evalrouter_contracts/validation.py,sha256=TzoAbtSETNdCgS_TvEjkAFVPJl3wv9eg-aD7IPKVrck,13085
|
|
9
|
+
kimpton_evalrouter/__init__.py,sha256=aoBQS7zNMjMnqx0QIJLG1zPlI9dEWAahxOXxVmvoA3E,3351
|
|
10
|
+
kimpton_evalrouter/__main__.py,sha256=k1ocEWawweo1qCJWNFAAvyxz3tcY13dzvCenHszij30,48
|
|
11
|
+
kimpton_evalrouter/_account.py,sha256=RfCLuPsebYrsUNMqMHdexYyz-xxOPM5C9eKtWDtvaiY,55036
|
|
12
|
+
kimpton_evalrouter/_boundary.py,sha256=VcHrxpjzWZzOjKdt-kb2uF5KcJMfnKzDLUTXtp_IW_U,7273
|
|
13
|
+
kimpton_evalrouter/_guided.py,sha256=Txz-6FuhJc0SD_O4a1AzdRBI3e7NS3wy_xNqVEOr_ZA,72581
|
|
14
|
+
kimpton_evalrouter/_helpers.py,sha256=LC3_lTqkxzM3qsO_lJ3lDx20tssuXBKr1UBxriz2ACY,3513
|
|
15
|
+
kimpton_evalrouter/_listing.py,sha256=gdyczD_cdGZk_Zs0ioPCJ8RNt5rBZlD8v43TW8_T8x0,40495
|
|
16
|
+
kimpton_evalrouter/_resources.py,sha256=agOtIrPjFaTOCPNV10OL9JX9lhm53-AWCkdsiwMWYcE,16919
|
|
17
|
+
kimpton_evalrouter/_transport.py,sha256=zvXixX2PIWWz67u_PfeksrzRafKU2KIJdVkT6H6oR_I,16033
|
|
18
|
+
kimpton_evalrouter/cli.py,sha256=ptw7wYOPVe6-hHayGaKZwbmdeQOTQJ6-BnFWjVdrD90,70185
|
|
19
|
+
kimpton_evalrouter/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
20
|
+
kimpton_evalrouter/types.py,sha256=5VZr73u2FzCxlCP3ZxehMSBIUuCyeC90TbDKlCT0EvU,29716
|
|
21
|
+
kimpton_evalrouter/render/__init__.py,sha256=80_zJZa19y9kQC9ZY7gGZQnDj36MBJNfGU8s2W-E81E,3765
|
|
22
|
+
kimpton_evalrouter/render/_console.py,sha256=4SN0ZBHc5rWoo0bSkDUEiBU_OKWgLifR0mFp4kqF1s8,3078
|
|
23
|
+
kimpton_evalrouter/render/_theme.py,sha256=sDu6JvztJkSfaSxnKxfhgQ8niww5tsYA27QFdTSRoVE,2238
|
|
24
|
+
kimpton_evalrouter/render/_widgets.py,sha256=aTRwFCI-aZOYSU6cpNSRinlfwfjr3ih9wVPYWK5RnzQ,4739
|
|
25
|
+
kimpton_evalrouter/render/live.py,sha256=6ZCDptTzpMAVqtICiljKlCU00tq0sJNLrqvtv78r7Gs,21431
|
|
26
|
+
kimpton_evalrouter/render/picker.py,sha256=OzaPIBfQUKSfa9x2u6u2EdbVjmrwePuN11aIC2ZlXRE,7625
|
|
27
|
+
kimpton_evalrouter/render/screens.py,sha256=vzo_g8V7aG6qO-ns_JBQej8Wh9mp5eKbngm8xXntUeA,32045
|
|
28
|
+
kimpton_evalrouter/render/tables.py,sha256=1s4g6yeBQ_CjBe9nu66rmb97NYDehPJMkFYEJU50FlE,14832
|
|
29
|
+
evalrouter-0.3.0.dist-info/METADATA,sha256=3Ie5EsgZ3W3kuuWxyoS9_qN_B33mQ0znkIDnWh7p8kY,12360
|
|
30
|
+
evalrouter-0.3.0.dist-info/WHEEL,sha256=qtCwoSJWgHk21S1Kb4ihdzI2rlJ1ZKaIurTj_ngOhyQ,87
|
|
31
|
+
evalrouter-0.3.0.dist-info/entry_points.txt,sha256=KHK6JH7M7ucqxHMqB3d2Rns9HJ3z-s4Yo_wlf90aHrU,59
|
|
32
|
+
evalrouter-0.3.0.dist-info/licenses/LICENSE,sha256=WsVi5Xlfp5XKnJGvoSNy722NnnWyv0Zp0p55gCZDgsU,1608
|
|
33
|
+
evalrouter-0.3.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
Kimpton EvalRouter SDK License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Kimpton. All rights reserved.
|
|
4
|
+
|
|
5
|
+
Permission is granted to download, install, execute, and make internal copies of
|
|
6
|
+
this SDK and its documentation solely to access the EvalRouter service and to
|
|
7
|
+
integrate that access into applications you own or control, subject to your
|
|
8
|
+
applicable EvalRouter service agreement and authorization.
|
|
9
|
+
|
|
10
|
+
This is proprietary software, not an open-source license. Except for the limited
|
|
11
|
+
permission above or as required by applicable law, no permission is granted to
|
|
12
|
+
modify, redistribute, sublicense, sell, or use this software to provide a
|
|
13
|
+
competing service. Preserve this license and all copyright notices in permitted
|
|
14
|
+
copies. All rights not expressly granted are reserved.
|
|
15
|
+
|
|
16
|
+
This license covers only the distributed SDK, CLI, and accompanying
|
|
17
|
+
documentation. It grants no rights to EvalRouter backend or service source,
|
|
18
|
+
benchmarks, datasets, models, credentials, trademarks, or other excluded material.
|
|
19
|
+
Third-party dependencies remain subject to their own licenses.
|
|
20
|
+
|
|
21
|
+
Downloading this software does not create an account, grant service access,
|
|
22
|
+
provide credits, or waive service charges. Access remains subject to separate
|
|
23
|
+
account, workspace, authorization, and billing requirements.
|
|
24
|
+
|
|
25
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
26
|
+
IMPLIED, INCLUDING WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR
|
|
27
|
+
PURPOSE, AND NONINFRINGEMENT. TO THE MAXIMUM EXTENT PERMITTED BY APPLICABLE LAW,
|
|
28
|
+
THE COPYRIGHT HOLDERS SHALL NOT BE LIABLE FOR ANY CLAIM, DAMAGES, OR OTHER
|
|
29
|
+
LIABILITY ARISING FROM THE SOFTWARE OR ITS USE.
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""Normative package-v1 contracts: models, canonicalization and offline validation.
|
|
2
|
+
|
|
3
|
+
This package is the one authoritative validator. The public `evalrouter`
|
|
4
|
+
distribution bundles it (its `[benchmark]` extra supplies pydantic), and the
|
|
5
|
+
internal `kimpton-evalrouter-contracts` workspace package serves the private
|
|
6
|
+
`kimpton-evalrouter-authoring` tools and the EvalRouter API. It reads bytes,
|
|
7
|
+
computes digests and refuses invalid packages; it never imports contributed
|
|
8
|
+
code, opens a network connection, signs anything or carries operator state.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
__version__ = "0.3.0"
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""Deterministic ZIPs of declared files only; exclusive publication after validation."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import tempfile
|
|
5
|
+
import zipfile
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from .validation import DirectoryReader, PackageError, canonical, manifest, validate
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def bundle(source: Path, destination: Path) -> dict:
|
|
12
|
+
reader = DirectoryReader(source)
|
|
13
|
+
package = manifest(reader)
|
|
14
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
15
|
+
fd, temporary = tempfile.mkstemp(prefix=".evalrouter-bundle-", dir=destination.parent)
|
|
16
|
+
os.close(fd)
|
|
17
|
+
temp = Path(temporary)
|
|
18
|
+
try:
|
|
19
|
+
with zipfile.ZipFile(temp, "w", compression=zipfile.ZIP_STORED) as archive:
|
|
20
|
+
for name in sorted(["evalrouter.json", *(file.path for file in package.files)]):
|
|
21
|
+
entry = zipfile.ZipInfo(name, date_time=(1980, 1, 1, 0, 0, 0))
|
|
22
|
+
entry.create_system = 3
|
|
23
|
+
entry.external_attr = 0o100644 << 16
|
|
24
|
+
with archive.open(entry, "w") as output:
|
|
25
|
+
if name == "evalrouter.json":
|
|
26
|
+
output.write(canonical(package.model_dump()))
|
|
27
|
+
else:
|
|
28
|
+
limit = next(f.size_bytes for f in package.files if f.path == name)
|
|
29
|
+
with reader.open(name) as stream:
|
|
30
|
+
copied = 0
|
|
31
|
+
while chunk := stream.read(min(1024 * 1024, limit + 1 - copied)):
|
|
32
|
+
copied += len(chunk)
|
|
33
|
+
if copied > limit:
|
|
34
|
+
raise PackageError(
|
|
35
|
+
f"{name}: file grew beyond its declared size"
|
|
36
|
+
)
|
|
37
|
+
output.write(chunk)
|
|
38
|
+
# Validate the actual immutable snapshot, not merely the source folder.
|
|
39
|
+
report = validate(temp)
|
|
40
|
+
try:
|
|
41
|
+
os.link(temp, destination)
|
|
42
|
+
except FileExistsError as exc:
|
|
43
|
+
raise PackageError("Bundle destination already exists; choose a new path") from exc
|
|
44
|
+
return report
|
|
45
|
+
finally:
|
|
46
|
+
temp.unlink(missing_ok=True)
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
"""`evalrouter.native-components.v1`: the typed component graph a native package points at.
|
|
2
|
+
|
|
3
|
+
Native packages only. `parameters` exists solely on `NativeProtocol`, so a
|
|
4
|
+
declarative `text.v1` package and a `koliseum` original package have nowhere to
|
|
5
|
+
carry the pointer and are out of scope for this contract.
|
|
6
|
+
|
|
7
|
+
The pointer rides `protocol.parameters` rather than a new envelope field because
|
|
8
|
+
`package_sha256` is `sha256(canonical(package.model_dump()))` with no
|
|
9
|
+
`exclude_defaults`: adding even an optional defaulted field to
|
|
10
|
+
`EvaluationPackage` would move the digest of every historical package.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import hashlib
|
|
14
|
+
import re
|
|
15
|
+
from typing import Annotated, Literal
|
|
16
|
+
|
|
17
|
+
from pydantic import AfterValidator, Field, model_validator
|
|
18
|
+
|
|
19
|
+
from .models import Contract, Digest, PackagePath, Slug, safe_path
|
|
20
|
+
|
|
21
|
+
CONTRACT = "evalrouter.native-components.v1"
|
|
22
|
+
POINTER_KEYS = ("component_contract", "component_path", "component_sha256")
|
|
23
|
+
INDEX_LIMIT = 256 * 1024
|
|
24
|
+
|
|
25
|
+
#: Critical extensions this build understands. An unrecognized critical key is a
|
|
26
|
+
#: refusal before preparation, quote, or execution; unknown optional keys
|
|
27
|
+
#: round-trip and contribute nothing to compiled semantics.
|
|
28
|
+
RECOGNIZED_CRITICAL_EXTENSIONS: frozenset[str] = frozenset()
|
|
29
|
+
|
|
30
|
+
#: Model operations a component may request, mapped onto declared capabilities.
|
|
31
|
+
OPERATION_CAPABILITY = {
|
|
32
|
+
"generate": "text_generation",
|
|
33
|
+
"continuation_loglikelihood": "continuation_loglikelihood",
|
|
34
|
+
}
|
|
35
|
+
Operation = Literal["generate", "continuation_loglikelihood"]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def entrypoint_target(value: str) -> str:
|
|
39
|
+
"""Split `<declared package path>:<symbol>` and return the path part."""
|
|
40
|
+
path, separator, symbol = value.rpartition(":")
|
|
41
|
+
if not separator or not path or not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", symbol):
|
|
42
|
+
raise ValueError("Entrypoints are <declared code path>:<symbol>")
|
|
43
|
+
return safe_path(path)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def checked_entrypoint(value: str) -> str:
|
|
47
|
+
entrypoint_target(value)
|
|
48
|
+
return value
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
Entrypoint = Annotated[
|
|
52
|
+
str,
|
|
53
|
+
Field(pattern=r"^[a-zA-Z0-9_./:-]+$", max_length=200),
|
|
54
|
+
AfterValidator(checked_entrypoint),
|
|
55
|
+
]
|
|
56
|
+
ExtensionName = Annotated[
|
|
57
|
+
str, Field(pattern=r"^[a-z][a-z0-9]*(?:[.-][a-z0-9]+)*\.v[1-9][0-9]*$", max_length=120)
|
|
58
|
+
]
|
|
59
|
+
ExtensionKey = Annotated[str, Field(pattern=r"^[a-z][a-z0-9_]*$", max_length=80)]
|
|
60
|
+
#: An extension body is arbitrary JSON so that an unknown *optional* extension
|
|
61
|
+
#: written by a later build round-trips byte-exactly through this one instead of
|
|
62
|
+
#: refusing the whole package. Object keys stay snake_case at every level, which
|
|
63
|
+
#: is the only shape restriction; pydantic-core's own recursion guard bounds
|
|
64
|
+
#: depth, and `INDEX_LIMIT` bounds total size.
|
|
65
|
+
type ExtensionValue = (
|
|
66
|
+
str | int | float | bool | None | list[ExtensionValue] | dict[ExtensionKey, ExtensionValue]
|
|
67
|
+
)
|
|
68
|
+
ExtensionBody = dict[ExtensionKey, ExtensionValue]
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def canonical_bytes(value: dict) -> bytes:
|
|
72
|
+
"""Canonical JSON bytes.
|
|
73
|
+
|
|
74
|
+
`canonical` is imported lazily because `validation` imports this module and
|
|
75
|
+
`packages/koliseum-benchmarks` imports `canonical` from `validation`, where
|
|
76
|
+
it must keep living.
|
|
77
|
+
"""
|
|
78
|
+
from .validation import canonical
|
|
79
|
+
|
|
80
|
+
return canonical(value)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def definition_digest(definition: dict) -> str:
|
|
84
|
+
"""SHA-256 over the canonical bytes of a component's `definition` only."""
|
|
85
|
+
return hashlib.sha256(canonical_bytes(definition)).hexdigest()
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class TasksDefinition(Contract):
|
|
89
|
+
#: `Literal[False]` rather than `bool`: `binding` is `Literal["package-dataset"]`
|
|
90
|
+
#: in v1, so no `true` value can ever compile. Typing it here puts the refusal
|
|
91
|
+
#: at the component-document seam that `validate()` reports, instead of letting
|
|
92
|
+
#: `validate()` bless a package the compiler always rejects. The material-bound
|
|
93
|
+
#: case arrives with `MaterialReceipt` in a later contract version.
|
|
94
|
+
binding: Literal["package-dataset"]
|
|
95
|
+
material_required: Literal[False]
|
|
96
|
+
selection: Literal["all"]
|
|
97
|
+
task_id_field: str = Field(min_length=1, max_length=120)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class RunnerDefinition(Contract):
|
|
101
|
+
dependency_lock: PackagePath
|
|
102
|
+
entrypoint: Entrypoint
|
|
103
|
+
framework: str = Field(min_length=1, max_length=80)
|
|
104
|
+
framework_version: str = Field(min_length=1, max_length=40)
|
|
105
|
+
kind: Literal["inspect", "lm-eval", "sandbox"]
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class ModelInterfaceDefinition(Contract):
|
|
109
|
+
operations: list[Operation] = Field(min_length=1, max_length=8)
|
|
110
|
+
|
|
111
|
+
@model_validator(mode="after")
|
|
112
|
+
def unique(self):
|
|
113
|
+
if len(set(self.operations)) != len(self.operations):
|
|
114
|
+
raise ValueError("Model interface operations must be unique")
|
|
115
|
+
return self
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class GraderDefinition(Contract):
|
|
119
|
+
aggregation: Literal["mean", "native"]
|
|
120
|
+
entrypoint: Entrypoint
|
|
121
|
+
metric: Slug
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class Component(Contract):
|
|
125
|
+
"""Wrapper/definition split: the digest covers the definition, never itself."""
|
|
126
|
+
|
|
127
|
+
sha256: Digest
|
|
128
|
+
|
|
129
|
+
@model_validator(mode="after")
|
|
130
|
+
def digest(self):
|
|
131
|
+
if definition_digest(self.definition.model_dump()) != self.sha256:
|
|
132
|
+
raise ValueError("Component sha256 must equal the digest of its definition")
|
|
133
|
+
return self
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class TasksComponent(Component):
|
|
137
|
+
definition: TasksDefinition
|
|
138
|
+
schema_version: Literal["evalrouter.component.tasks.v1"]
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class RunnerComponent(Component):
|
|
142
|
+
definition: RunnerDefinition
|
|
143
|
+
schema_version: Literal["evalrouter.component.runner.v1"]
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
class ModelInterfaceComponent(Component):
|
|
147
|
+
definition: ModelInterfaceDefinition
|
|
148
|
+
schema_version: Literal["evalrouter.component.model-interface.v1"]
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
class GraderComponent(Component):
|
|
152
|
+
definition: GraderDefinition
|
|
153
|
+
schema_version: Literal["evalrouter.component.grader.v1"]
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
class ComponentSet(Contract):
|
|
157
|
+
"""Keyed by family name, so duplicates are structurally impossible."""
|
|
158
|
+
|
|
159
|
+
grader: GraderComponent
|
|
160
|
+
model_interface: ModelInterfaceComponent
|
|
161
|
+
runner: RunnerComponent
|
|
162
|
+
tasks: TasksComponent
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
class Extensions(Contract):
|
|
166
|
+
"""Both maps are required: an omitted section is not the same as an empty one."""
|
|
167
|
+
|
|
168
|
+
critical: dict[ExtensionName, ExtensionBody] = Field(max_length=50)
|
|
169
|
+
optional: dict[ExtensionName, ExtensionBody] = Field(max_length=50)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
class NativeComponentsIndex(Contract):
|
|
173
|
+
components: ComponentSet
|
|
174
|
+
extensions: Extensions
|
|
175
|
+
schema_version: Literal["evalrouter.native-components.v1"]
|
|
176
|
+
|
|
177
|
+
def digest(self) -> str:
|
|
178
|
+
return hashlib.sha256(canonical_bytes(self.model_dump())).hexdigest()
|
|
179
|
+
|
|
180
|
+
def unrecognized_critical(self) -> list[str]:
|
|
181
|
+
return sorted(set(self.extensions.critical) - RECOGNIZED_CRITICAL_EXTENSIONS)
|