evalrouter 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. evalrouter-0.3.0/LICENSE +29 -0
  2. evalrouter-0.3.0/PKG-INFO +267 -0
  3. evalrouter-0.3.0/README.md +252 -0
  4. evalrouter-0.3.0/pyproject.toml +34 -0
  5. evalrouter-0.3.0/src/evalrouter_contracts/__init__.py +11 -0
  6. evalrouter-0.3.0/src/evalrouter_contracts/bundle.py +46 -0
  7. evalrouter-0.3.0/src/evalrouter_contracts/components.py +181 -0
  8. evalrouter-0.3.0/src/evalrouter_contracts/models.py +288 -0
  9. evalrouter-0.3.0/src/evalrouter_contracts/protocol.py +221 -0
  10. evalrouter-0.3.0/src/evalrouter_contracts/py.typed +0 -0
  11. evalrouter-0.3.0/src/evalrouter_contracts/templates.py +136 -0
  12. evalrouter-0.3.0/src/evalrouter_contracts/validation.py +329 -0
  13. evalrouter-0.3.0/src/kimpton_evalrouter/__init__.py +112 -0
  14. evalrouter-0.3.0/src/kimpton_evalrouter/__main__.py +3 -0
  15. evalrouter-0.3.0/src/kimpton_evalrouter/_account.py +1379 -0
  16. evalrouter-0.3.0/src/kimpton_evalrouter/_boundary.py +191 -0
  17. evalrouter-0.3.0/src/kimpton_evalrouter/_guided.py +1889 -0
  18. evalrouter-0.3.0/src/kimpton_evalrouter/_helpers.py +101 -0
  19. evalrouter-0.3.0/src/kimpton_evalrouter/_listing.py +1107 -0
  20. evalrouter-0.3.0/src/kimpton_evalrouter/_resources.py +537 -0
  21. evalrouter-0.3.0/src/kimpton_evalrouter/_transport.py +436 -0
  22. evalrouter-0.3.0/src/kimpton_evalrouter/cli.py +1676 -0
  23. evalrouter-0.3.0/src/kimpton_evalrouter/py.typed +0 -0
  24. evalrouter-0.3.0/src/kimpton_evalrouter/render/__init__.py +99 -0
  25. evalrouter-0.3.0/src/kimpton_evalrouter/render/_console.py +91 -0
  26. evalrouter-0.3.0/src/kimpton_evalrouter/render/_theme.py +80 -0
  27. evalrouter-0.3.0/src/kimpton_evalrouter/render/_widgets.py +132 -0
  28. evalrouter-0.3.0/src/kimpton_evalrouter/render/live.py +514 -0
  29. evalrouter-0.3.0/src/kimpton_evalrouter/render/picker.py +230 -0
  30. evalrouter-0.3.0/src/kimpton_evalrouter/render/screens.py +842 -0
  31. evalrouter-0.3.0/src/kimpton_evalrouter/render/tables.py +427 -0
  32. evalrouter-0.3.0/src/kimpton_evalrouter/types.py +1171 -0
@@ -0,0 +1,29 @@
1
+ Kimpton EvalRouter SDK License
2
+
3
+ Copyright (c) 2026 Kimpton. All rights reserved.
4
+
5
+ Permission is granted to download, install, execute, and make internal copies of
6
+ this SDK and its documentation solely to access the EvalRouter service and to
7
+ integrate that access into applications you own or control, subject to your
8
+ applicable EvalRouter service agreement and authorization.
9
+
10
+ This is proprietary software, not an open-source license. Except for the limited
11
+ permission above or as required by applicable law, no permission is granted to
12
+ modify, redistribute, sublicense, sell, or use this software to provide a
13
+ competing service. Preserve this license and all copyright notices in permitted
14
+ copies. All rights not expressly granted are reserved.
15
+
16
+ This license covers only the distributed SDK, CLI, and accompanying
17
+ documentation. It grants no rights to EvalRouter backend or service source,
18
+ benchmarks, datasets, models, credentials, trademarks, or other excluded material.
19
+ Third-party dependencies remain subject to their own licenses.
20
+
21
+ Downloading this software does not create an account, grant service access,
22
+ provide credits, or waive service charges. Access remains subject to separate
23
+ account, workspace, authorization, and billing requirements.
24
+
25
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
26
+ IMPLIED, INCLUDING WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR
27
+ PURPOSE, AND NONINFRINGEMENT. TO THE MAXIMUM EXTENT PERMITTED BY APPLICABLE LAW,
28
+ THE COPYRIGHT HOLDERS SHALL NOT BE LIABLE FOR ANY CLAIM, DAMAGES, OR OTHER
29
+ LIABILITY ARISING FROM THE SOFTWARE OR ITS USE.
@@ -0,0 +1,267 @@
1
+ Metadata-Version: 2.4
2
+ Name: evalrouter
3
+ Version: 0.3.0
4
+ Summary: Typed Kimpton evaluation client and command-line interface
5
+ License-Expression: LicenseRef-Kimpton-EvalRouter-SDK
6
+ License-File: LICENSE
7
+ Requires-Python: <3.14,>=3.12
8
+ Requires-Dist: httpx==0.28.1
9
+ Requires-Dist: rich==15.0.0
10
+ Provides-Extra: benchmark
11
+ Requires-Dist: pydantic==2.13.5; extra == 'benchmark'
12
+ Provides-Extra: keyring
13
+ Requires-Dist: keyring==25.7.0; extra == 'keyring'
14
+ Description-Content-Type: text/markdown
15
+
16
+ # EvalRouter CLI
17
+
18
+ Evaluate models or qualified repository agents from your terminal: discover
19
+ supported targets, review a spending cap, run an evaluation, and export results. Requires Python 3.12
20
+ or 3.13. Distributed under the proprietary [Kimpton EvalRouter SDK License](LICENSE).
21
+
22
+ ## Install and create your account
23
+
24
+ We recommend [uv](https://docs.astral.sh/uv/getting-started/installation/) to install the CLI in its own Python environment:
25
+
26
+ ```sh
27
+ uv tool install --python 3.12 evalrouter
28
+ evalrouter --help
29
+ ```
30
+
31
+ With pip, use `pip install evalrouter`. uv supplies Python 3.12 if needed. If
32
+ your shell cannot find `evalrouter`, run `uv tool update-shell` and restart your
33
+ terminal. Upgrade an existing installation with `uv tool upgrade evalrouter`.
34
+
35
+ This package was previously published as `kimpton-evalrouter-sdk`. The command,
36
+ the `kimpton_evalrouter` import and every API are unchanged. To switch, run
37
+ `uv tool uninstall kimpton-evalrouter-sdk` (or
38
+ `pip uninstall -y kimpton-evalrouter-sdk evalrouter`) and install `evalrouter`.
39
+
40
+ The package supplies the `evalrouter` command; no repository checkout is needed.
41
+ [Create an EvalRouter account](https://evalrouter.ai/register), verify your email,
42
+ and create a workspace key in [API key settings](https://evalrouter.ai/settings/api-keys).
43
+ Package installation does not grant account access or evaluation credit.
44
+
45
+ Set `EVALROUTER_API_KEY` and `EVALROUTER_WORKSPACE_ID` through your secret manager
46
+ or private environment. Use `https://api.evalrouter.ai` for `EVALROUTER_BASE_URL`,
47
+ without `/v1`. Never put credentials in arguments, URLs, request files or logs.
48
+ The CLI uses an existing workspace key; browser sign-up is a separate step.
49
+
50
+ ## Run an evaluation in one command
51
+
52
+ ```sh
53
+ evalrouter run gpqa-diamond --model gpt-4o-mini
54
+ ```
55
+
56
+ `run` matches the benchmark and model by slug, name or ID, suggesting the
57
+ closest names after a typo. It creates a free quote, shows the coverage,
58
+ conservative estimate, spending cap and concurrency, and asks `Run it? [Y/n]`.
59
+ Only a yes starts paid work. It then follows progress and prints the scores,
60
+ the change from your previous completed run of the same benchmark and model,
61
+ and the report link. In a terminal, missing choices are asked for; run
62
+ `evalrouter run` alone to pick everything. `--sample N` or `--full` sets
63
+ coverage, `--max-cost USD` the cap and `--dry-run` stops after the quote.
64
+
65
+ Without a terminal (CI, agents, `--json` or `--no-input`) `run` never prompts
66
+ and refuses unless both `--yes` and `--max-cost` are given:
67
+
68
+ ```sh
69
+ evalrouter run gpqa-diamond --model gpt-4o-mini --max-cost 5 --yes --json
70
+ ```
71
+
72
+ The idempotency key of each accepted quote is saved locally before the run is
73
+ submitted (`~/.local/state/evalrouter`, or `EVALROUTER_STATE_DIR`; identifiers
74
+ only, never credentials). Repeating the same command after a dropped connection
75
+ resumes that run instead of starting another; `--new` starts a separate one.
76
+ Afterwards, `results`, `export --format html,json`, `wait` and `open` default to
77
+ your last run and accept a unique run ID prefix. Watching a run survives brief
78
+ API outages such as a 503 during a deployment.
79
+
80
+ ## Lists
81
+
82
+ `catalog`, `catalog --models`, `status`, `connections list` and `workspace list`
83
+ print a compact table sized to your terminal with one suggested next command;
84
+ `catalog` hides benchmarks in review or unavailable unless you pass `--all`, and
85
+ `catalog -i` browses them interactively. Every list accepts `--json`,
86
+ `--jq '.data[].id'` (a small jq subset), `-w/--wide`, `--columns a,b`,
87
+ `--sort COL` (`--sort=-COL` descending), `--filter KEY=VALUE` and `--limit N`.
88
+ Piped output stays JSON. Colour follows `NO_COLOR`, `CI` and `TERM=dumb`.
89
+
90
+ `run` shows the quote (expected cost from your previous run when known, worst
91
+ case, and the smallest workable cap) before asking for the cap, refuses a cap
92
+ that cannot fit one task, and asks before repeating a run that just finished
93
+ (scripts pass `--new`). `--all-metrics` lists every metric and comparison.
94
+
95
+ ## Discover, quote, then run
96
+
97
+ ```sh
98
+ evalrouter catalog --status active --json
99
+ evalrouter catalog --models --json
100
+ evalrouter catalog --slug BENCHMARK_SLUG --json
101
+ ```
102
+
103
+ Choose a profile whose `quote_availability.status` is `ready_for_quote` and a
104
+ compatible managed model route from those responses. An active catalog entry
105
+ alone does not mean it is ready to run.
106
+ Save this as `quote.json`, replacing both uppercase identifiers:
107
+
108
+ ```json
109
+ {
110
+ "model": {"kind": "managed", "route_id": "MODEL_ROUTE_ID"},
111
+ "selection": {"profile_ids": ["BENCHMARK_PROFILE_ID"]},
112
+ "coverage": {"mode": "sample", "sample_count": 3, "seed": 42},
113
+ "max_charge_microusd": "1000000"
114
+ }
115
+ ```
116
+
117
+ ```sh
118
+ evalrouter quote --config quote.json --json
119
+ ```
120
+
121
+ Review compatibility, coverage, cost components, warnings and expiry. A quote
122
+ starts no paid work. The example's $1 platform cap is not a price guarantee or
123
+ promise that a particular evaluation fits. Paid runs require available credit.
124
+
125
+ Persist the returned quote ID and an operation key before submitting:
126
+
127
+ ```sh
128
+ evalrouter run --quote REVIEWED_QUOTE_ID --idempotency-key SAVED_OPERATION_KEY --json
129
+ evalrouter wait RUN_ID --wait-timeout 3600 --json
130
+ evalrouter results RUN_ID --json
131
+ evalrouter export RUN_ID --format json --output result.json
132
+ evalrouter export RUN_ID --format csv --output result.csv
133
+ evalrouter export RUN_ID --format html --output result.html
134
+ ```
135
+
136
+ Replace `RUN_ID` with the returned ID. A lost submission response is recovered
137
+ with the same quote and operation key; a new key may start separate work.
138
+ Inspect terminal status, coverage, errors and billing alongside scores. A sample
139
+ is not a full-benchmark score. Exports refuse overwrite by default; use the
140
+ result's integer `--version` for repeatable reports.
141
+
142
+ ## Supported commands
143
+
144
+ `catalog` (`--models`, `--slug` or `--environment`), `connections list/create/check/update/disable`,
145
+ `quote`, `run`, `status`, `wait`, `cancel`, `results`, `export` and `open`.
146
+ Repository qualification uses `agent-builds options/preview/create/list/status/cancel`.
147
+ Authoring your own benchmark package uses `benchmark init/validate/bundle`.
148
+ Run each command with `--help` for exact flags. A connected model's credential
149
+ is never a command argument: `connections create` asks with hidden input in a
150
+ terminal, and scripts use `--key-stdin` or `--key-env NAME`
151
+ (`connections update --rotate-key` replaces it). Connection checks can send
152
+ a small model request, and your model provider may bill separately.
153
+
154
+ For `wait` and `run --wait`, `--progress auto` shows elapsed time and status in
155
+ an interactive terminal, with a percentage when validated processed and planned
156
+ counts are available. Use `--progress plain` for structured progress or
157
+ `--progress off` to suppress it. Redirected stderr and `--json` retain structured
158
+ progress. A progress percentage counts processed samples; it is not a score.
159
+
160
+ Use `--config -` for JSON on stdin. `--json` writes one result on stdout and
161
+ progress on stderr. Exit codes: 0 success; 1 API/transport or failed run;
162
+ 2 invalid input or rejected request (including a refused unattended `run`);
163
+ 4 partial/cancelled waited run; 5 confirmation declined, nothing started. Interrupting
164
+ a local wait does not cancel server work. Use `evalrouter cancel RUN_ID`
165
+ explicitly when cancellation is intended.
166
+
167
+ See the [CLI documentation](https://evalrouter.ai/developers/docs) for account
168
+ setup, model selection, spending semantics, recovery and versioned exports.
169
+ The CLI uses the [model API](https://evalrouter.ai/developers/docs/model-api),
170
+ which is also documented for direct HTTP integrations.
171
+
172
+ ## Author your own benchmark
173
+
174
+ `benchmark init`, `benchmark validate` and `benchmark bundle` run entirely on
175
+ your machine: no account, no network request, and nothing from your package is
176
+ imported or executed. The normative validator ships inside `evalrouter`, and
177
+ its one dependency (pydantic) is an optional extra so the base installation
178
+ stays thin (httpx and rich only). With pip, use `pip install 'evalrouter[benchmark]'`:
179
+
180
+ ```sh
181
+ uv tool install --python 3.12 "evalrouter[benchmark]"
182
+ evalrouter benchmark init ./my-benchmark --namespace example --name my-benchmark
183
+ evalrouter benchmark validate ./my-benchmark
184
+ evalrouter benchmark bundle ./my-benchmark --output my-benchmark.zip
185
+ ```
186
+
187
+ Without the extra these three commands stop before reading your package and
188
+ report `benchmark_extra_required` with the exact install string; `--help` and
189
+ every network command keep working. `init` writes an explicitly synthetic draft
190
+ and claims no license: replace the tasks, rights, references and limits before
191
+ submitting. Local validation is not platform admission — EvalRouter revalidates
192
+ every submission on its own servers.
193
+
194
+ ## Repository agents
195
+
196
+ Repository commands require version 0.2.0 or later and a service that has enabled
197
+ repository qualification for the workspace. Installing the package does not
198
+ enable that service. Check its current availability and supported targets first:
199
+
200
+ ```sh
201
+ evalrouter agent-builds options --json
202
+ evalrouter catalog --environment ENVIRONMENT_FAMILY_PATH --json
203
+ ```
204
+
205
+ The catalog lookup takes the family path before `@`; qualification inputs still
206
+ require the exact versioned reference returned by options. Stop if qualification
207
+ is disabled or the desired exact target/model is absent.
208
+ The supported runtime is locked Node/npm with an `evalrouter-agent.json` manifest;
209
+ custom images and Python agent runtimes are not supported. Save `build.json`:
210
+
211
+ ```json
212
+ {
213
+ "repository_url": "YOUR_PUBLIC_GITHUB_REPOSITORY_URL",
214
+ "ref": "main",
215
+ "manifest_path": "evalrouter-agent.json",
216
+ "qualification": {
217
+ "environment_ref": "EXACT_ENVIRONMENT_FROM_OPTIONS",
218
+ "model": {"kind": "managed", "route_id": "MODEL_FROM_OPTIONS"}
219
+ },
220
+ "max_cost_microusd": "1000000"
221
+ }
222
+ ```
223
+
224
+ For a private repository, connect GitHub in the web application and explicitly
225
+ select that repository for this workspace. Add
226
+ `source: {"github_connection_id": "CONNECTION_ID", "github_repository_id": 123}`
227
+ using the two actual IDs from that connection. These IDs select a grant; they
228
+ are not credentials. The CLI does not accept GitHub tokens or authorize access.
229
+
230
+ ```sh
231
+ evalrouter agent-builds preview --config build.json --json
232
+ ```
233
+
234
+ Preview resolves an immutable commit and returns the qualification scope, costs,
235
+ cap and expiry. It creates no job or credit hold and starts no sandbox/model
236
+ work. Review it, then save the quote ID and a durable operation key before:
237
+
238
+ ```sh
239
+ evalrouter agent-builds create --quote REVIEWED_BUILD_QUOTE --idempotency-key SAVED_BUILD_KEY --json
240
+ evalrouter agent-builds status BUILD_JOB_ID --json
241
+ evalrouter agent-builds list --quote REVIEWED_BUILD_QUOTE --json
242
+ ```
243
+
244
+ Creation reserves the quoted cap and may incur its stated costs. Recover an
245
+ unknown submission with the same quote and key or the quote-filtered list;
246
+ a new key may create new work. Cancel explicitly with
247
+ `evalrouter agent-builds cancel BUILD_JOB_ID`. A `cancelling` or `reconciling`
248
+ status remains unresolved. Wait for terminal cleanup and zero held funds.
249
+
250
+ Ready requires `status: ready`, `cleanup_confirmed: true`, zero held funds and
251
+ an `agent.ref`. It establishes only the returned scope, not benchmark quality.
252
+ Use that record's exact `qualification.environment_ref`, `qualification.split`
253
+ and `qualification.evaluation_coverage` in a separate evaluation quote:
254
+
255
+ ```json
256
+ {
257
+ "agent": "READY_AGENT_REF",
258
+ "selection": {"environment": "RETURNED_ENVIRONMENT_REF", "split": "RETURNED_SPLIT"},
259
+ "coverage": "REPLACE_WITH_RETURNED_EVALUATION_COVERAGE_OBJECT",
260
+ "max_charge_microusd": "1000000"
261
+ }
262
+ ```
263
+
264
+ Replace the coverage placeholder with the entire returned object; do not guess
265
+ tasks, counts or seeds. Use the existing `quote`, `run`, `wait`, `results` and
266
+ `export` commands above. Qualification and evaluation have separate caps and
267
+ charges. A qualification pass does not submit an evaluation automatically.
@@ -0,0 +1,252 @@
1
+ # EvalRouter CLI
2
+
3
+ Evaluate models or qualified repository agents from your terminal: discover
4
+ supported targets, review a spending cap, run an evaluation, and export results. Requires Python 3.12
5
+ or 3.13. Distributed under the proprietary [Kimpton EvalRouter SDK License](LICENSE).
6
+
7
+ ## Install and create your account
8
+
9
+ We recommend [uv](https://docs.astral.sh/uv/getting-started/installation/) to install the CLI in its own Python environment:
10
+
11
+ ```sh
12
+ uv tool install --python 3.12 evalrouter
13
+ evalrouter --help
14
+ ```
15
+
16
+ With pip, use `pip install evalrouter`. uv supplies Python 3.12 if needed. If
17
+ your shell cannot find `evalrouter`, run `uv tool update-shell` and restart your
18
+ terminal. Upgrade an existing installation with `uv tool upgrade evalrouter`.
19
+
20
+ This package was previously published as `kimpton-evalrouter-sdk`. The command,
21
+ the `kimpton_evalrouter` import and every API are unchanged. To switch, run
22
+ `uv tool uninstall kimpton-evalrouter-sdk` (or
23
+ `pip uninstall -y kimpton-evalrouter-sdk evalrouter`) and install `evalrouter`.
24
+
25
+ The package supplies the `evalrouter` command; no repository checkout is needed.
26
+ [Create an EvalRouter account](https://evalrouter.ai/register), verify your email,
27
+ and create a workspace key in [API key settings](https://evalrouter.ai/settings/api-keys).
28
+ Package installation does not grant account access or evaluation credit.
29
+
30
+ Set `EVALROUTER_API_KEY` and `EVALROUTER_WORKSPACE_ID` through your secret manager
31
+ or private environment. Use `https://api.evalrouter.ai` for `EVALROUTER_BASE_URL`,
32
+ without `/v1`. Never put credentials in arguments, URLs, request files or logs.
33
+ The CLI uses an existing workspace key; browser sign-up is a separate step.
34
+
35
+ ## Run an evaluation in one command
36
+
37
+ ```sh
38
+ evalrouter run gpqa-diamond --model gpt-4o-mini
39
+ ```
40
+
41
+ `run` matches the benchmark and model by slug, name or ID, suggesting the
42
+ closest names after a typo. It creates a free quote, shows the coverage,
43
+ conservative estimate, spending cap and concurrency, and asks `Run it? [Y/n]`.
44
+ Only a yes starts paid work. It then follows progress and prints the scores,
45
+ the change from your previous completed run of the same benchmark and model,
46
+ and the report link. In a terminal, missing choices are asked for; run
47
+ `evalrouter run` alone to pick everything. `--sample N` or `--full` sets
48
+ coverage, `--max-cost USD` the cap and `--dry-run` stops after the quote.
49
+
50
+ Without a terminal (CI, agents, `--json` or `--no-input`) `run` never prompts
51
+ and refuses unless both `--yes` and `--max-cost` are given:
52
+
53
+ ```sh
54
+ evalrouter run gpqa-diamond --model gpt-4o-mini --max-cost 5 --yes --json
55
+ ```
56
+
57
+ The idempotency key of each accepted quote is saved locally before the run is
58
+ submitted (`~/.local/state/evalrouter`, or `EVALROUTER_STATE_DIR`; identifiers
59
+ only, never credentials). Repeating the same command after a dropped connection
60
+ resumes that run instead of starting another; `--new` starts a separate one.
61
+ Afterwards, `results`, `export --format html,json`, `wait` and `open` default to
62
+ your last run and accept a unique run ID prefix. Watching a run survives brief
63
+ API outages such as a 503 during a deployment.
64
+
65
+ ## Lists
66
+
67
+ `catalog`, `catalog --models`, `status`, `connections list` and `workspace list`
68
+ print a compact table sized to your terminal with one suggested next command;
69
+ `catalog` hides benchmarks in review or unavailable unless you pass `--all`, and
70
+ `catalog -i` browses them interactively. Every list accepts `--json`,
71
+ `--jq '.data[].id'` (a small jq subset), `-w/--wide`, `--columns a,b`,
72
+ `--sort COL` (`--sort=-COL` descending), `--filter KEY=VALUE` and `--limit N`.
73
+ Piped output stays JSON. Colour follows `NO_COLOR`, `CI` and `TERM=dumb`.
74
+
75
+ `run` shows the quote (expected cost from your previous run when known, worst
76
+ case, and the smallest workable cap) before asking for the cap, refuses a cap
77
+ that cannot fit one task, and asks before repeating a run that just finished
78
+ (scripts pass `--new`). `--all-metrics` lists every metric and comparison.
79
+
80
+ ## Discover, quote, then run
81
+
82
+ ```sh
83
+ evalrouter catalog --status active --json
84
+ evalrouter catalog --models --json
85
+ evalrouter catalog --slug BENCHMARK_SLUG --json
86
+ ```
87
+
88
+ Choose a profile whose `quote_availability.status` is `ready_for_quote` and a
89
+ compatible managed model route from those responses. An active catalog entry
90
+ alone does not mean it is ready to run.
91
+ Save this as `quote.json`, replacing both uppercase identifiers:
92
+
93
+ ```json
94
+ {
95
+ "model": {"kind": "managed", "route_id": "MODEL_ROUTE_ID"},
96
+ "selection": {"profile_ids": ["BENCHMARK_PROFILE_ID"]},
97
+ "coverage": {"mode": "sample", "sample_count": 3, "seed": 42},
98
+ "max_charge_microusd": "1000000"
99
+ }
100
+ ```
101
+
102
+ ```sh
103
+ evalrouter quote --config quote.json --json
104
+ ```
105
+
106
+ Review compatibility, coverage, cost components, warnings and expiry. A quote
107
+ starts no paid work. The example's $1 platform cap is not a price guarantee or
108
+ promise that a particular evaluation fits. Paid runs require available credit.
109
+
110
+ Persist the returned quote ID and an operation key before submitting:
111
+
112
+ ```sh
113
+ evalrouter run --quote REVIEWED_QUOTE_ID --idempotency-key SAVED_OPERATION_KEY --json
114
+ evalrouter wait RUN_ID --wait-timeout 3600 --json
115
+ evalrouter results RUN_ID --json
116
+ evalrouter export RUN_ID --format json --output result.json
117
+ evalrouter export RUN_ID --format csv --output result.csv
118
+ evalrouter export RUN_ID --format html --output result.html
119
+ ```
120
+
121
+ Replace `RUN_ID` with the returned ID. A lost submission response is recovered
122
+ with the same quote and operation key; a new key may start separate work.
123
+ Inspect terminal status, coverage, errors and billing alongside scores. A sample
124
+ is not a full-benchmark score. Exports refuse overwrite by default; use the
125
+ result's integer `--version` for repeatable reports.
126
+
127
+ ## Supported commands
128
+
129
+ `catalog` (`--models`, `--slug` or `--environment`), `connections list/create/check/update/disable`,
130
+ `quote`, `run`, `status`, `wait`, `cancel`, `results`, `export` and `open`.
131
+ Repository qualification uses `agent-builds options/preview/create/list/status/cancel`.
132
+ Authoring your own benchmark package uses `benchmark init/validate/bundle`.
133
+ Run each command with `--help` for exact flags. A connected model's credential
134
+ is never a command argument: `connections create` asks with hidden input in a
135
+ terminal, and scripts use `--key-stdin` or `--key-env NAME`
136
+ (`connections update --rotate-key` replaces it). Connection checks can send
137
+ a small model request, and your model provider may bill separately.
138
+
139
+ For `wait` and `run --wait`, `--progress auto` shows elapsed time and status in
140
+ an interactive terminal, with a percentage when validated processed and planned
141
+ counts are available. Use `--progress plain` for structured progress or
142
+ `--progress off` to suppress it. Redirected stderr and `--json` retain structured
143
+ progress. A progress percentage counts processed samples; it is not a score.
144
+
145
+ Use `--config -` for JSON on stdin. `--json` writes one result on stdout and
146
+ progress on stderr. Exit codes: 0 success; 1 API/transport or failed run;
147
+ 2 invalid input or rejected request (including a refused unattended `run`);
148
+ 4 partial/cancelled waited run; 5 confirmation declined, nothing started. Interrupting
149
+ a local wait does not cancel server work. Use `evalrouter cancel RUN_ID`
150
+ explicitly when cancellation is intended.
151
+
152
+ See the [CLI documentation](https://evalrouter.ai/developers/docs) for account
153
+ setup, model selection, spending semantics, recovery and versioned exports.
154
+ The CLI uses the [model API](https://evalrouter.ai/developers/docs/model-api),
155
+ which is also documented for direct HTTP integrations.
156
+
157
+ ## Author your own benchmark
158
+
159
+ `benchmark init`, `benchmark validate` and `benchmark bundle` run entirely on
160
+ your machine: no account, no network request, and nothing from your package is
161
+ imported or executed. The normative validator ships inside `evalrouter`, and
162
+ its one dependency (pydantic) is an optional extra so the base installation
163
+ stays thin (httpx and rich only). With pip, use `pip install 'evalrouter[benchmark]'`:
164
+
165
+ ```sh
166
+ uv tool install --python 3.12 "evalrouter[benchmark]"
167
+ evalrouter benchmark init ./my-benchmark --namespace example --name my-benchmark
168
+ evalrouter benchmark validate ./my-benchmark
169
+ evalrouter benchmark bundle ./my-benchmark --output my-benchmark.zip
170
+ ```
171
+
172
+ Without the extra these three commands stop before reading your package and
173
+ report `benchmark_extra_required` with the exact install string; `--help` and
174
+ every network command keep working. `init` writes an explicitly synthetic draft
175
+ and claims no license: replace the tasks, rights, references and limits before
176
+ submitting. Local validation is not platform admission — EvalRouter revalidates
177
+ every submission on its own servers.
178
+
179
+ ## Repository agents
180
+
181
+ Repository commands require version 0.2.0 or later and a service that has enabled
182
+ repository qualification for the workspace. Installing the package does not
183
+ enable that service. Check its current availability and supported targets first:
184
+
185
+ ```sh
186
+ evalrouter agent-builds options --json
187
+ evalrouter catalog --environment ENVIRONMENT_FAMILY_PATH --json
188
+ ```
189
+
190
+ The catalog lookup takes the family path before `@`; qualification inputs still
191
+ require the exact versioned reference returned by options. Stop if qualification
192
+ is disabled or the desired exact target/model is absent.
193
+ The supported runtime is locked Node/npm with an `evalrouter-agent.json` manifest;
194
+ custom images and Python agent runtimes are not supported. Save `build.json`:
195
+
196
+ ```json
197
+ {
198
+ "repository_url": "YOUR_PUBLIC_GITHUB_REPOSITORY_URL",
199
+ "ref": "main",
200
+ "manifest_path": "evalrouter-agent.json",
201
+ "qualification": {
202
+ "environment_ref": "EXACT_ENVIRONMENT_FROM_OPTIONS",
203
+ "model": {"kind": "managed", "route_id": "MODEL_FROM_OPTIONS"}
204
+ },
205
+ "max_cost_microusd": "1000000"
206
+ }
207
+ ```
208
+
209
+ For a private repository, connect GitHub in the web application and explicitly
210
+ select that repository for this workspace. Add
211
+ `source: {"github_connection_id": "CONNECTION_ID", "github_repository_id": 123}`
212
+ using the two actual IDs from that connection. These IDs select a grant; they
213
+ are not credentials. The CLI does not accept GitHub tokens or authorize access.
214
+
215
+ ```sh
216
+ evalrouter agent-builds preview --config build.json --json
217
+ ```
218
+
219
+ Preview resolves an immutable commit and returns the qualification scope, costs,
220
+ cap and expiry. It creates no job or credit hold and starts no sandbox/model
221
+ work. Review it, then save the quote ID and a durable operation key before:
222
+
223
+ ```sh
224
+ evalrouter agent-builds create --quote REVIEWED_BUILD_QUOTE --idempotency-key SAVED_BUILD_KEY --json
225
+ evalrouter agent-builds status BUILD_JOB_ID --json
226
+ evalrouter agent-builds list --quote REVIEWED_BUILD_QUOTE --json
227
+ ```
228
+
229
+ Creation reserves the quoted cap and may incur its stated costs. Recover an
230
+ unknown submission with the same quote and key or the quote-filtered list;
231
+ a new key may create new work. Cancel explicitly with
232
+ `evalrouter agent-builds cancel BUILD_JOB_ID`. A `cancelling` or `reconciling`
233
+ status remains unresolved. Wait for terminal cleanup and zero held funds.
234
+
235
+ Ready requires `status: ready`, `cleanup_confirmed: true`, zero held funds and
236
+ an `agent.ref`. It establishes only the returned scope, not benchmark quality.
237
+ Use that record's exact `qualification.environment_ref`, `qualification.split`
238
+ and `qualification.evaluation_coverage` in a separate evaluation quote:
239
+
240
+ ```json
241
+ {
242
+ "agent": "READY_AGENT_REF",
243
+ "selection": {"environment": "RETURNED_ENVIRONMENT_REF", "split": "RETURNED_SPLIT"},
244
+ "coverage": "REPLACE_WITH_RETURNED_EVALUATION_COVERAGE_OBJECT",
245
+ "max_charge_microusd": "1000000"
246
+ }
247
+ ```
248
+
249
+ Replace the coverage placeholder with the entire returned object; do not guess
250
+ tasks, counts or seeds. Use the existing `quote`, `run`, `wait`, `results` and
251
+ `export` commands above. Qualification and evaluation have separate caps and
252
+ charges. A qualification pass does not submit an evaluation automatically.
@@ -0,0 +1,34 @@
1
+ [build-system]
2
+ requires = ["hatchling==1.27.0"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "evalrouter"
7
+ version = "0.3.0"
8
+ description = "Typed Kimpton evaluation client and command-line interface"
9
+ readme = "README.md"
10
+ license = "LicenseRef-Kimpton-EvalRouter-SDK"
11
+ license-files = ["LICENSE"]
12
+ requires-python = ">=3.12,<3.14"
13
+ # The base installation is the HTTP client plus terminal rendering: httpx for
14
+ # the API and rich (pure Python) for the CLI's human output. Both are exact
15
+ # pins; rich 15.0.0 matches deploy/experiments/requirements.lock. Validation
16
+ # stays out: the published wheel bundles the normative `evalrouter_contracts`
17
+ # modules, but their one dependency (pydantic) arrives only through the
18
+ # optional `[benchmark]` extra, never through this list.
19
+ dependencies = ["httpx==0.28.1", "rich==15.0.0"]
20
+
21
+ [project.optional-dependencies]
22
+ # Exactly the internal contracts package's own dependency list
23
+ # (packages/contracts/pyproject.toml); build_packages.py refuses any drift.
24
+ benchmark = ["pydantic==2.13.5"]
25
+ # Account sign-in stores tokens in the OS keychain when this extra is
26
+ # installed; otherwise it uses a private 0600 file. Kept optional so the base
27
+ # installation and every service/runner lock stay unchanged.
28
+ keyring = ["keyring==25.7.0"]
29
+
30
+ [project.scripts]
31
+ evalrouter = "kimpton_evalrouter.cli:main"
32
+
33
+ [tool.hatch.build.targets.wheel]
34
+ packages = ["src/kimpton_evalrouter", "src/evalrouter_contracts"]
@@ -0,0 +1,11 @@
1
+ """Normative package-v1 contracts: models, canonicalization and offline validation.
2
+
3
+ This package is the one authoritative validator. The public `evalrouter`
4
+ distribution bundles it (its `[benchmark]` extra supplies pydantic), and the
5
+ internal `kimpton-evalrouter-contracts` workspace package serves the private
6
+ `kimpton-evalrouter-authoring` tools and the EvalRouter API. It reads bytes,
7
+ computes digests and refuses invalid packages; it never imports contributed
8
+ code, opens a network connection, signs anything or carries operator state.
9
+ """
10
+
11
+ __version__ = "0.3.0"
@@ -0,0 +1,46 @@
1
+ """Deterministic ZIPs of declared files only; exclusive publication after validation."""
2
+
3
+ import os
4
+ import tempfile
5
+ import zipfile
6
+ from pathlib import Path
7
+
8
+ from .validation import DirectoryReader, PackageError, canonical, manifest, validate
9
+
10
+
11
+ def bundle(source: Path, destination: Path) -> dict:
12
+ reader = DirectoryReader(source)
13
+ package = manifest(reader)
14
+ destination.parent.mkdir(parents=True, exist_ok=True)
15
+ fd, temporary = tempfile.mkstemp(prefix=".evalrouter-bundle-", dir=destination.parent)
16
+ os.close(fd)
17
+ temp = Path(temporary)
18
+ try:
19
+ with zipfile.ZipFile(temp, "w", compression=zipfile.ZIP_STORED) as archive:
20
+ for name in sorted(["evalrouter.json", *(file.path for file in package.files)]):
21
+ entry = zipfile.ZipInfo(name, date_time=(1980, 1, 1, 0, 0, 0))
22
+ entry.create_system = 3
23
+ entry.external_attr = 0o100644 << 16
24
+ with archive.open(entry, "w") as output:
25
+ if name == "evalrouter.json":
26
+ output.write(canonical(package.model_dump()))
27
+ else:
28
+ limit = next(f.size_bytes for f in package.files if f.path == name)
29
+ with reader.open(name) as stream:
30
+ copied = 0
31
+ while chunk := stream.read(min(1024 * 1024, limit + 1 - copied)):
32
+ copied += len(chunk)
33
+ if copied > limit:
34
+ raise PackageError(
35
+ f"{name}: file grew beyond its declared size"
36
+ )
37
+ output.write(chunk)
38
+ # Validate the actual immutable snapshot, not merely the source folder.
39
+ report = validate(temp)
40
+ try:
41
+ os.link(temp, destination)
42
+ except FileExistsError as exc:
43
+ raise PackageError("Bundle destination already exists; choose a new path") from exc
44
+ return report
45
+ finally:
46
+ temp.unlink(missing_ok=True)