rhylthyme-cli-runner 0.2.0a0__tar.gz → 0.2.2a0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rhylthyme_cli_runner-0.2.0a0/src/rhylthyme_cli_runner.egg-info → rhylthyme_cli_runner-0.2.2a0}/PKG-INFO +154 -3
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/README.md +153 -2
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/setup.py +1 -1
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/__init__.py +1 -1
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/cli.py +4 -1
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/llm.py +191 -4
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/program_planner.py +20 -4
- rhylthyme_cli_runner-0.2.2a0/src/rhylthyme_cli_runner/remote/checks.py +663 -0
- rhylthyme_cli_runner-0.2.2a0/src/rhylthyme_cli_runner/remote/cli.py +594 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/remote/mcp_client.py +69 -16
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0/src/rhylthyme_cli_runner.egg-info}/PKG-INFO +154 -3
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner.egg-info/SOURCES.txt +5 -0
- rhylthyme_cli_runner-0.2.2a0/tests/test_eval_providers.py +292 -0
- rhylthyme_cli_runner-0.2.2a0/tests/test_mcp_checks.py +441 -0
- rhylthyme_cli_runner-0.2.2a0/tests/test_planner_time_strings.py +20 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_remote.py +95 -0
- rhylthyme_cli_runner-0.2.2a0/tests/test_skill.py +71 -0
- rhylthyme_cli_runner-0.2.0a0/src/rhylthyme_cli_runner/remote/cli.py +0 -282
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/LICENSE +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/pyproject.toml +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/setup.cfg +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/cli.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/environment_icons.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/environment_loader.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/environment_schemas.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/__init__.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/compare.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/gold.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/harness.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/matcher.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/metrics.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/__init__.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/authoring_guide.md +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/baseline.md +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/baseline.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/extract.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/four_turn.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/report.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/expand_replicates.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/__init__.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/calibrate.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/calibrate_cli.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/cli.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/evaluate.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/evaluate_cli.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/factors.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/hash.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/predict.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/recorder.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/render.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/replay.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/report.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/report_cli.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/store.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/synth.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/usable.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/instance_checks.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/program_runner.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/remote/__init__.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/remote/auth.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/validate_program.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner.egg-info/dependency_links.txt +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner.egg-info/entry_points.txt +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner.egg-info/requires.txt +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner.egg-info/top_level.txt +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_calibrate.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_cli.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_eval_baseline.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_eval_harness.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_eval_scorer.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_evaluate.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_example_validation.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_examples_integration.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_expand_replicates.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_hash_parity.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_history_report.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_predict.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_program_runner.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_replay.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_run_recorder.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_synth.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_usable.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_validate_examples_ci.py +0 -0
- {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_validate_program.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rhylthyme-cli-runner
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2a0
|
|
4
4
|
Summary: CLI runner for Rhylthyme real-time program schedules
|
|
5
5
|
Home-page: https://github.com/rhylthyme/rhylthyme-cli-runner
|
|
6
6
|
Author: Rhylthyme Team
|
|
@@ -197,9 +197,30 @@ rhylthyme run examples/programs/breakfast_schedule.json --environment kitchen
|
|
|
197
197
|
rhylthyme run examples/programs/breakfast_schedule.json --no-validate
|
|
198
198
|
```
|
|
199
199
|
|
|
200
|
+
### Analyze and Publish a Program
|
|
201
|
+
|
|
202
|
+
Neither needs an account; both are computed by the hosted MCP server.
|
|
203
|
+
|
|
204
|
+
```bash
|
|
205
|
+
# Total length, critical path, resource conflicts
|
|
206
|
+
rhylthyme analyze examples/programs/breakfast_schedule.json
|
|
207
|
+
|
|
208
|
+
# When does each step start if breakfast is at 8:30?
|
|
209
|
+
rhylthyme analyze examples/programs/breakfast_schedule.json --finish-at 8:30am
|
|
210
|
+
|
|
211
|
+
# A live, shareable timeline with timers; prints the URL
|
|
212
|
+
rhylthyme publish examples/programs/breakfast_schedule.json
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
`validate` checks structure only. Two steps that want the only oven at the
|
|
216
|
+
same time pass validation; `analyze` reports them, and `--strict` makes that
|
|
217
|
+
a non-zero exit.
|
|
218
|
+
|
|
200
219
|
### Optimize a Program
|
|
201
220
|
|
|
202
|
-
|
|
221
|
+
`plan` is an older stagger heuristic that reads the pre-0.2 program format; on
|
|
222
|
+
programs written with `stepId` and `task` (all current examples) it writes the
|
|
223
|
+
program back unchanged. Use `analyze` to find contention.
|
|
203
224
|
|
|
204
225
|
```bash
|
|
205
226
|
# Optimize a program and save to new file
|
|
@@ -227,6 +248,29 @@ rhylthyme validate-environments
|
|
|
227
248
|
rhylthyme environment-info kitchen
|
|
228
249
|
```
|
|
229
250
|
|
|
251
|
+
## Claude Skill
|
|
252
|
+
|
|
253
|
+
[`skills/rhylthyme`](skills/rhylthyme) is an [Agent Skill](https://docs.claude.com/en/docs/agents-and-tools/agent-skills/overview)
|
|
254
|
+
that teaches Claude (Claude Code, the Claude apps, the Agent SDK) to turn a
|
|
255
|
+
protocol, recipe or run sheet into a validated program with this CLI, check
|
|
256
|
+
it for conflicts, and hand back a live timeline. It follows the layout of
|
|
257
|
+
[K-Dense scientific skills](https://github.com/K-Dense-AI/claude-scientific-skills):
|
|
258
|
+
a `SKILL.md` plus `references/`.
|
|
259
|
+
|
|
260
|
+
```bash
|
|
261
|
+
# Claude Code, for one user
|
|
262
|
+
mkdir -p ~/.claude/skills
|
|
263
|
+
cp -r skills/rhylthyme ~/.claude/skills/
|
|
264
|
+
|
|
265
|
+
# or for one project
|
|
266
|
+
mkdir -p .claude/skills && cp -r skills/rhylthyme .claude/skills/
|
|
267
|
+
```
|
|
268
|
+
|
|
269
|
+
Then ask, for example, "time a Western blot so imaging is at 4 pm" or "two PCR
|
|
270
|
+
protocols, one thermocycler: when do I start each?". `tests/test_skill.py`
|
|
271
|
+
validates every program in the skill and checks that every command it names
|
|
272
|
+
exists.
|
|
273
|
+
|
|
230
274
|
## Program File Examples
|
|
231
275
|
|
|
232
276
|
### Simple Breakfast Schedule
|
|
@@ -305,6 +349,39 @@ Environment: `RHYLTHYME_TOKEN` (access token, overrides the stored session),
|
|
|
305
349
|
`RHYLTHYME_MCP_URL` (default `https://mcp.rhylthyme.com/mcp`),
|
|
306
350
|
`RHYLTHYME_SITE_URL` (default `https://www.rhylthyme.com`).
|
|
307
351
|
|
|
352
|
+
### `rhylthyme mcp-test`
|
|
353
|
+
|
|
354
|
+
Smoke-tests a Rhylthyme MCP server and exits 1 if any check fails. By
|
|
355
|
+
default it runs the read-only checks against all five hosted endpoints
|
|
356
|
+
(`/mcp`, `/kitchen/mcp`, `/lab/mcp`, `/events/mcp`, `/gym/mcp`):
|
|
357
|
+
|
|
358
|
+
| Check | What it proves |
|
|
359
|
+
|---|---|
|
|
360
|
+
| `initialize` | server name, protocol version, tools/resources/prompts capabilities, instructions |
|
|
361
|
+
| `tools` | core tools and the endpoint's one-shot tools are listed, each with a description and schema |
|
|
362
|
+
| `validate-good` / `validate-bad` | a valid program passes (including a `type: "compound"` trigger); a dangling step reference is rejected with a fix hint |
|
|
363
|
+
| `analyze` | makespan, critical path and wall-clock itinerary for a known program |
|
|
364
|
+
| `resources` / `prompts` | schema and guides are readable, a bundled example validates, `plan_schedule` substitutes its arguments |
|
|
365
|
+
| `json-accept` | clients that do not accept SSE (`*/*`, `application/json`) get JSON, not 406 |
|
|
366
|
+
| `bad-requests` | unknown method gives -32601, unknown tool gives an error, never a 5xx |
|
|
367
|
+
| `login-gate` | account tools refuse without a token and point at `login` |
|
|
368
|
+
| `catalog` | public search returns entries that load with a URL (empty catalog = warning) |
|
|
369
|
+
| `publish` (`--publish`) | `visualize_schedule` returns a URL on the right site; the page and PNG load |
|
|
370
|
+
| `generate` (`--generate`) | the model-backed `import_text` returns a program that validates (needs `login`) |
|
|
371
|
+
|
|
372
|
+
```bash
|
|
373
|
+
rhylthyme mcp-test # everything read-only
|
|
374
|
+
rhylthyme mcp-test -e lab --publish # one endpoint, plus a real share
|
|
375
|
+
rhylthyme mcp-test -k catalog -k tools # only some checks
|
|
376
|
+
rhylthyme mcp-test --url http://localhost:3000/mcp -e generic
|
|
377
|
+
rhylthyme mcp-test --json --strict # cron / CI: warnings fail too
|
|
378
|
+
```
|
|
379
|
+
|
|
380
|
+
Requests carry `mcp-test` in the User-Agent so the hosted server logs the
|
|
381
|
+
errors these checks provoke without alerting anyone. The same suite is
|
|
382
|
+
importable (`rhylthyme_cli_runner.remote.checks.run_suite`), and
|
|
383
|
+
`RHYLTHYME_MCP_LIVE=1 pytest tests/test_mcp_checks.py` runs it live.
|
|
384
|
+
|
|
308
385
|
### `rhylthyme login` / `logout` / `whoami`
|
|
309
386
|
|
|
310
387
|
`login` opens the rhylthyme.com sign-in page and receives the session on a
|
|
@@ -331,9 +408,33 @@ Runs programs with interactive terminal UI.
|
|
|
331
408
|
- `--validate / --no-validate`: Validate before running (default: True)
|
|
332
409
|
- `--auto-start`: Automatically start without manual trigger
|
|
333
410
|
|
|
411
|
+
### `rhylthyme analyze`
|
|
412
|
+
|
|
413
|
+
Total length, critical path, what gates each link of it, resource conflicts
|
|
414
|
+
and tracks that finish early. No sign-in; nothing is published.
|
|
415
|
+
|
|
416
|
+
**Options:**
|
|
417
|
+
- `--finish-at TEXT`: when everything must be finished (`19:00`, `7:30pm` or ISO 8601); prints local clock times per step
|
|
418
|
+
- `--start-at TEXT`: when the program starts; ignored with `--finish-at`
|
|
419
|
+
- `--strict`: exit non-zero when there are resource conflicts
|
|
420
|
+
- `--json`: the full analysis
|
|
421
|
+
|
|
422
|
+
### `rhylthyme publish`
|
|
423
|
+
|
|
424
|
+
Publishes a program file as a live timeline and prints its URL. No sign-in.
|
|
425
|
+
A published timeline is reachable by anyone who has the link.
|
|
426
|
+
|
|
427
|
+
**Options:**
|
|
428
|
+
- `-e, --env`: `generic`, `kitchen`, `lab`, `events`, `gym` (default: from `environmentType`)
|
|
429
|
+
- `-q, --quiet`: print only the URL
|
|
430
|
+
- `--json`: `url`, `shareId`, `imageUrl`, `makespanSeconds`, `warnings`
|
|
431
|
+
- `--open`: open it in a browser
|
|
432
|
+
|
|
334
433
|
### `rhylthyme plan`
|
|
335
434
|
|
|
336
|
-
|
|
435
|
+
An older stagger heuristic. It reads the pre-0.2 program format and leaves
|
|
436
|
+
programs written with `stepId` and `task` unchanged; use `analyze` to find
|
|
437
|
+
contention and edit the triggers.
|
|
337
438
|
|
|
338
439
|
**Options:**
|
|
339
440
|
- `--verbose, -v`: Show detailed planning information
|
|
@@ -614,6 +715,56 @@ make test-unit # excludes llm
|
|
|
614
715
|
RHYLTHYME_EVAL_LIVE=1 pytest -m llm # one real call, opt in
|
|
615
716
|
```
|
|
616
717
|
|
|
718
|
+
|
|
719
|
+
### Other models and a spending cap
|
|
720
|
+
|
|
721
|
+
`--model` also accepts models served through an OpenAI-compatible API. The
|
|
722
|
+
provider is picked from the model id and the key is read from the
|
|
723
|
+
environment:
|
|
724
|
+
|
|
725
|
+
| Model id | Provider | Key |
|
|
726
|
+
|---|---|---|
|
|
727
|
+
| `claude-*` | Anthropic | `ANTHROPIC_API_KEY` |
|
|
728
|
+
| `deepseek-*` (e.g. `deepseek-flash`, `deepseek-v4-pro`) | api.deepseek.com | `DEEPSEEK_API_KEY` |
|
|
729
|
+
| `qwen*` | Alibaba DashScope (international) | `DASHSCOPE_API_KEY` |
|
|
730
|
+
| `kimi-*`, `moonshot-*` | Moonshot | `MOONSHOT_API_KEY` |
|
|
731
|
+
| `glm-*` | Z.ai | `ZAI_API_KEY` |
|
|
732
|
+
| anything with a slash, e.g. `deepseek/deepseek-flash` | OpenRouter | `OPENROUTER_API_KEY` |
|
|
733
|
+
| any id, with `RHYLTHYME_EVAL_BASE_URL` set | that endpoint (vLLM, Ollama, ...) | `RHYLTHYME_EVAL_API_KEY`, optional for localhost |
|
|
734
|
+
|
|
735
|
+
Costs come from the price table in `eval/llm.py` (DeepSeek at its peak
|
|
736
|
+
rates, so estimates are upper bounds). For a model the table does not know,
|
|
737
|
+
set `RHYLTHYME_EVAL_PRICE_IN` and `RHYLTHYME_EVAL_PRICE_OUT` in USD per
|
|
738
|
+
million tokens.
|
|
739
|
+
|
|
740
|
+
`eval/run_model_comparison.py` runs a model under a hard budget. It runs one
|
|
741
|
+
gold program at a time, both prompts per program, records each program's
|
|
742
|
+
cost in `eval/models/spend-ledger.json`, and stops before the next program
|
|
743
|
+
could cross `--cap`. The ledger is cumulative across models and
|
|
744
|
+
invocations, and the script refuses a model with no known price (it would
|
|
745
|
+
be recorded as $0 and the cap would never trip).
|
|
746
|
+
|
|
747
|
+
One OpenRouter key reaches OpenAI, Qwen and Meta models (ids with a vendor
|
|
748
|
+
prefix). Run the cheapest first and raise the cumulative cap as you go, so
|
|
749
|
+
no single model can spend the whole allowance. Reasoning models need a
|
|
750
|
+
higher per-call output ceiling than the default 16,000 tokens:
|
|
751
|
+
|
|
752
|
+
```bash
|
|
753
|
+
export OPENROUTER_API_KEY=...
|
|
754
|
+
L=eval/models/spend-ledger-openrouter.json
|
|
755
|
+
python eval/run_model_comparison.py --ledger $L --model meta-llama/llama-4-maverick --cap 0.80 --first-guess 0.05
|
|
756
|
+
python eval/run_model_comparison.py --ledger $L --model qwen/qwen3.8-flash --cap 2.00 --first-guess 0.08 --max-tokens 64000
|
|
757
|
+
python eval/run_model_comparison.py --ledger $L --model openai/gpt-5.6-luna --cap 4.50 --first-guess 0.15 --max-tokens 64000
|
|
758
|
+
python eval/compare_models.py
|
|
759
|
+
```
|
|
760
|
+
|
|
761
|
+
```bash
|
|
762
|
+
export DEEPSEEK_API_KEY=...
|
|
763
|
+
python eval/run_model_comparison.py --model deepseek-flash --cap 6.70 --dry-run
|
|
764
|
+
python eval/run_model_comparison.py --model deepseek-flash --cap 6.70
|
|
765
|
+
python eval/compare_models.py # like-for-like table across every model run so far
|
|
766
|
+
```
|
|
767
|
+
|
|
617
768
|
## Development
|
|
618
769
|
|
|
619
770
|
1. Clone the repository
|
|
@@ -137,9 +137,30 @@ rhylthyme run examples/programs/breakfast_schedule.json --environment kitchen
|
|
|
137
137
|
rhylthyme run examples/programs/breakfast_schedule.json --no-validate
|
|
138
138
|
```
|
|
139
139
|
|
|
140
|
+
### Analyze and Publish a Program
|
|
141
|
+
|
|
142
|
+
Neither needs an account; both are computed by the hosted MCP server.
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
# Total length, critical path, resource conflicts
|
|
146
|
+
rhylthyme analyze examples/programs/breakfast_schedule.json
|
|
147
|
+
|
|
148
|
+
# When does each step start if breakfast is at 8:30?
|
|
149
|
+
rhylthyme analyze examples/programs/breakfast_schedule.json --finish-at 8:30am
|
|
150
|
+
|
|
151
|
+
# A live, shareable timeline with timers; prints the URL
|
|
152
|
+
rhylthyme publish examples/programs/breakfast_schedule.json
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
`validate` checks structure only. Two steps that want the only oven at the
|
|
156
|
+
same time pass validation; `analyze` reports them, and `--strict` makes that
|
|
157
|
+
a non-zero exit.
|
|
158
|
+
|
|
140
159
|
### Optimize a Program
|
|
141
160
|
|
|
142
|
-
|
|
161
|
+
`plan` is an older stagger heuristic that reads the pre-0.2 program format; on
|
|
162
|
+
programs written with `stepId` and `task` (all current examples) it writes the
|
|
163
|
+
program back unchanged. Use `analyze` to find contention.
|
|
143
164
|
|
|
144
165
|
```bash
|
|
145
166
|
# Optimize a program and save to new file
|
|
@@ -167,6 +188,29 @@ rhylthyme validate-environments
|
|
|
167
188
|
rhylthyme environment-info kitchen
|
|
168
189
|
```
|
|
169
190
|
|
|
191
|
+
## Claude Skill
|
|
192
|
+
|
|
193
|
+
[`skills/rhylthyme`](skills/rhylthyme) is an [Agent Skill](https://docs.claude.com/en/docs/agents-and-tools/agent-skills/overview)
|
|
194
|
+
that teaches Claude (Claude Code, the Claude apps, the Agent SDK) to turn a
|
|
195
|
+
protocol, recipe or run sheet into a validated program with this CLI, check
|
|
196
|
+
it for conflicts, and hand back a live timeline. It follows the layout of
|
|
197
|
+
[K-Dense scientific skills](https://github.com/K-Dense-AI/claude-scientific-skills):
|
|
198
|
+
a `SKILL.md` plus `references/`.
|
|
199
|
+
|
|
200
|
+
```bash
|
|
201
|
+
# Claude Code, for one user
|
|
202
|
+
mkdir -p ~/.claude/skills
|
|
203
|
+
cp -r skills/rhylthyme ~/.claude/skills/
|
|
204
|
+
|
|
205
|
+
# or for one project
|
|
206
|
+
mkdir -p .claude/skills && cp -r skills/rhylthyme .claude/skills/
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
Then ask, for example, "time a Western blot so imaging is at 4 pm" or "two PCR
|
|
210
|
+
protocols, one thermocycler: when do I start each?". `tests/test_skill.py`
|
|
211
|
+
validates every program in the skill and checks that every command it names
|
|
212
|
+
exists.
|
|
213
|
+
|
|
170
214
|
## Program File Examples
|
|
171
215
|
|
|
172
216
|
### Simple Breakfast Schedule
|
|
@@ -245,6 +289,39 @@ Environment: `RHYLTHYME_TOKEN` (access token, overrides the stored session),
|
|
|
245
289
|
`RHYLTHYME_MCP_URL` (default `https://mcp.rhylthyme.com/mcp`),
|
|
246
290
|
`RHYLTHYME_SITE_URL` (default `https://www.rhylthyme.com`).
|
|
247
291
|
|
|
292
|
+
### `rhylthyme mcp-test`
|
|
293
|
+
|
|
294
|
+
Smoke-tests a Rhylthyme MCP server and exits 1 if any check fails. By
|
|
295
|
+
default it runs the read-only checks against all five hosted endpoints
|
|
296
|
+
(`/mcp`, `/kitchen/mcp`, `/lab/mcp`, `/events/mcp`, `/gym/mcp`):
|
|
297
|
+
|
|
298
|
+
| Check | What it proves |
|
|
299
|
+
|---|---|
|
|
300
|
+
| `initialize` | server name, protocol version, tools/resources/prompts capabilities, instructions |
|
|
301
|
+
| `tools` | core tools and the endpoint's one-shot tools are listed, each with a description and schema |
|
|
302
|
+
| `validate-good` / `validate-bad` | a valid program passes (including a `type: "compound"` trigger); a dangling step reference is rejected with a fix hint |
|
|
303
|
+
| `analyze` | makespan, critical path and wall-clock itinerary for a known program |
|
|
304
|
+
| `resources` / `prompts` | schema and guides are readable, a bundled example validates, `plan_schedule` substitutes its arguments |
|
|
305
|
+
| `json-accept` | clients that do not accept SSE (`*/*`, `application/json`) get JSON, not 406 |
|
|
306
|
+
| `bad-requests` | unknown method gives -32601, unknown tool gives an error, never a 5xx |
|
|
307
|
+
| `login-gate` | account tools refuse without a token and point at `login` |
|
|
308
|
+
| `catalog` | public search returns entries that load with a URL (empty catalog = warning) |
|
|
309
|
+
| `publish` (`--publish`) | `visualize_schedule` returns a URL on the right site; the page and PNG load |
|
|
310
|
+
| `generate` (`--generate`) | the model-backed `import_text` returns a program that validates (needs `login`) |
|
|
311
|
+
|
|
312
|
+
```bash
|
|
313
|
+
rhylthyme mcp-test # everything read-only
|
|
314
|
+
rhylthyme mcp-test -e lab --publish # one endpoint, plus a real share
|
|
315
|
+
rhylthyme mcp-test -k catalog -k tools # only some checks
|
|
316
|
+
rhylthyme mcp-test --url http://localhost:3000/mcp -e generic
|
|
317
|
+
rhylthyme mcp-test --json --strict # cron / CI: warnings fail too
|
|
318
|
+
```
|
|
319
|
+
|
|
320
|
+
Requests carry `mcp-test` in the User-Agent so the hosted server logs the
|
|
321
|
+
errors these checks provoke without alerting anyone. The same suite is
|
|
322
|
+
importable (`rhylthyme_cli_runner.remote.checks.run_suite`), and
|
|
323
|
+
`RHYLTHYME_MCP_LIVE=1 pytest tests/test_mcp_checks.py` runs it live.
|
|
324
|
+
|
|
248
325
|
### `rhylthyme login` / `logout` / `whoami`
|
|
249
326
|
|
|
250
327
|
`login` opens the rhylthyme.com sign-in page and receives the session on a
|
|
@@ -271,9 +348,33 @@ Runs programs with interactive terminal UI.
|
|
|
271
348
|
- `--validate / --no-validate`: Validate before running (default: True)
|
|
272
349
|
- `--auto-start`: Automatically start without manual trigger
|
|
273
350
|
|
|
351
|
+
### `rhylthyme analyze`
|
|
352
|
+
|
|
353
|
+
Total length, critical path, what gates each link of it, resource conflicts
|
|
354
|
+
and tracks that finish early. No sign-in; nothing is published.
|
|
355
|
+
|
|
356
|
+
**Options:**
|
|
357
|
+
- `--finish-at TEXT`: when everything must be finished (`19:00`, `7:30pm` or ISO 8601); prints local clock times per step
|
|
358
|
+
- `--start-at TEXT`: when the program starts; ignored with `--finish-at`
|
|
359
|
+
- `--strict`: exit non-zero when there are resource conflicts
|
|
360
|
+
- `--json`: the full analysis
|
|
361
|
+
|
|
362
|
+
### `rhylthyme publish`
|
|
363
|
+
|
|
364
|
+
Publishes a program file as a live timeline and prints its URL. No sign-in.
|
|
365
|
+
A published timeline is reachable by anyone who has the link.
|
|
366
|
+
|
|
367
|
+
**Options:**
|
|
368
|
+
- `-e, --env`: `generic`, `kitchen`, `lab`, `events`, `gym` (default: from `environmentType`)
|
|
369
|
+
- `-q, --quiet`: print only the URL
|
|
370
|
+
- `--json`: `url`, `shareId`, `imageUrl`, `makespanSeconds`, `warnings`
|
|
371
|
+
- `--open`: open it in a browser
|
|
372
|
+
|
|
274
373
|
### `rhylthyme plan`
|
|
275
374
|
|
|
276
|
-
|
|
375
|
+
An older stagger heuristic. It reads the pre-0.2 program format and leaves
|
|
376
|
+
programs written with `stepId` and `task` unchanged; use `analyze` to find
|
|
377
|
+
contention and edit the triggers.
|
|
277
378
|
|
|
278
379
|
**Options:**
|
|
279
380
|
- `--verbose, -v`: Show detailed planning information
|
|
@@ -554,6 +655,56 @@ make test-unit # excludes llm
|
|
|
554
655
|
RHYLTHYME_EVAL_LIVE=1 pytest -m llm # one real call, opt in
|
|
555
656
|
```
|
|
556
657
|
|
|
658
|
+
|
|
659
|
+
### Other models and a spending cap
|
|
660
|
+
|
|
661
|
+
`--model` also accepts models served through an OpenAI-compatible API. The
|
|
662
|
+
provider is picked from the model id and the key is read from the
|
|
663
|
+
environment:
|
|
664
|
+
|
|
665
|
+
| Model id | Provider | Key |
|
|
666
|
+
|---|---|---|
|
|
667
|
+
| `claude-*` | Anthropic | `ANTHROPIC_API_KEY` |
|
|
668
|
+
| `deepseek-*` (e.g. `deepseek-flash`, `deepseek-v4-pro`) | api.deepseek.com | `DEEPSEEK_API_KEY` |
|
|
669
|
+
| `qwen*` | Alibaba DashScope (international) | `DASHSCOPE_API_KEY` |
|
|
670
|
+
| `kimi-*`, `moonshot-*` | Moonshot | `MOONSHOT_API_KEY` |
|
|
671
|
+
| `glm-*` | Z.ai | `ZAI_API_KEY` |
|
|
672
|
+
| anything with a slash, e.g. `deepseek/deepseek-flash` | OpenRouter | `OPENROUTER_API_KEY` |
|
|
673
|
+
| any id, with `RHYLTHYME_EVAL_BASE_URL` set | that endpoint (vLLM, Ollama, ...) | `RHYLTHYME_EVAL_API_KEY`, optional for localhost |
|
|
674
|
+
|
|
675
|
+
Costs come from the price table in `eval/llm.py` (DeepSeek at its peak
|
|
676
|
+
rates, so estimates are upper bounds). For a model the table does not know,
|
|
677
|
+
set `RHYLTHYME_EVAL_PRICE_IN` and `RHYLTHYME_EVAL_PRICE_OUT` in USD per
|
|
678
|
+
million tokens.
|
|
679
|
+
|
|
680
|
+
`eval/run_model_comparison.py` runs a model under a hard budget. It runs one
|
|
681
|
+
gold program at a time, both prompts per program, records each program's
|
|
682
|
+
cost in `eval/models/spend-ledger.json`, and stops before the next program
|
|
683
|
+
could cross `--cap`. The ledger is cumulative across models and
|
|
684
|
+
invocations, and the script refuses a model with no known price (it would
|
|
685
|
+
be recorded as $0 and the cap would never trip).
|
|
686
|
+
|
|
687
|
+
One OpenRouter key reaches OpenAI, Qwen and Meta models (ids with a vendor
|
|
688
|
+
prefix). Run the cheapest first and raise the cumulative cap as you go, so
|
|
689
|
+
no single model can spend the whole allowance. Reasoning models need a
|
|
690
|
+
higher per-call output ceiling than the default 16,000 tokens:
|
|
691
|
+
|
|
692
|
+
```bash
|
|
693
|
+
export OPENROUTER_API_KEY=...
|
|
694
|
+
L=eval/models/spend-ledger-openrouter.json
|
|
695
|
+
python eval/run_model_comparison.py --ledger $L --model meta-llama/llama-4-maverick --cap 0.80 --first-guess 0.05
|
|
696
|
+
python eval/run_model_comparison.py --ledger $L --model qwen/qwen3.8-flash --cap 2.00 --first-guess 0.08 --max-tokens 64000
|
|
697
|
+
python eval/run_model_comparison.py --ledger $L --model openai/gpt-5.6-luna --cap 4.50 --first-guess 0.15 --max-tokens 64000
|
|
698
|
+
python eval/compare_models.py
|
|
699
|
+
```
|
|
700
|
+
|
|
701
|
+
```bash
|
|
702
|
+
export DEEPSEEK_API_KEY=...
|
|
703
|
+
python eval/run_model_comparison.py --model deepseek-flash --cap 6.70 --dry-run
|
|
704
|
+
python eval/run_model_comparison.py --model deepseek-flash --cap 6.70
|
|
705
|
+
python eval/compare_models.py # like-for-like table across every model run so far
|
|
706
|
+
```
|
|
707
|
+
|
|
557
708
|
## Development
|
|
558
709
|
|
|
559
710
|
1. Clone the repository
|
|
@@ -19,7 +19,7 @@ def read_readme():
|
|
|
19
19
|
|
|
20
20
|
setup(
|
|
21
21
|
name="rhylthyme-cli-runner",
|
|
22
|
-
version="0.2.
|
|
22
|
+
version="0.2.2a0",
|
|
23
23
|
description="CLI runner for Rhylthyme real-time program schedules",
|
|
24
24
|
long_description=read_readme(),
|
|
25
25
|
long_description_content_type="text/markdown",
|
{rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/__init__.py
RENAMED
|
@@ -6,7 +6,7 @@ This package provides the command-line interface for running and validating
|
|
|
6
6
|
Rhylthyme real-time program schedules.
|
|
7
7
|
"""
|
|
8
8
|
|
|
9
|
-
__version__ = "0.2.
|
|
9
|
+
__version__ = "0.2.2a0"
|
|
10
10
|
__author__ = "Rhylthyme Team"
|
|
11
11
|
__description__ = "CLI runner for Rhylthyme real-time program schedules"
|
|
12
12
|
|
{rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/cli.py
RENAMED
|
@@ -321,7 +321,10 @@ def _live(
|
|
|
321
321
|
if fake_client_path:
|
|
322
322
|
with open(fake_client_path, "r", encoding="utf-8") as handle:
|
|
323
323
|
fake_responses = json.load(handle)
|
|
324
|
-
|
|
324
|
+
try:
|
|
325
|
+
client = make_client(fake_responses, model=model)
|
|
326
|
+
except RuntimeError as exc:
|
|
327
|
+
raise click.UsageError(str(exc))
|
|
325
328
|
|
|
326
329
|
gold_set = _load_gold(gold_dir)
|
|
327
330
|
out_root = Path(out_dir)
|
{rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/llm.py
RENAMED
|
@@ -16,7 +16,12 @@ Plus a small price table so runs can log an estimated cost.
|
|
|
16
16
|
|
|
17
17
|
from __future__ import annotations
|
|
18
18
|
|
|
19
|
+
import json
|
|
20
|
+
import os
|
|
19
21
|
import threading
|
|
22
|
+
import time
|
|
23
|
+
import urllib.error
|
|
24
|
+
import urllib.request
|
|
20
25
|
from dataclasses import dataclass, field
|
|
21
26
|
from typing import Any, Dict, List, Mapping, Optional, Protocol, Sequence, Union
|
|
22
27
|
|
|
@@ -38,9 +43,42 @@ PRICES_PER_MTOK: Dict[str, tuple] = {
|
|
|
38
43
|
"claude-sonnet-5": (2.00, 10.00),
|
|
39
44
|
"claude-sonnet-4-6": (3.00, 15.00),
|
|
40
45
|
"claude-haiku-4-5": (1.00, 5.00),
|
|
46
|
+
# DeepSeek, from api-docs.deepseek.com/quick_start/pricing on 2026-09-19.
|
|
47
|
+
# PEAK rates (off-peak is half), so a budget computed from these is an
|
|
48
|
+
# upper bound. Cache-miss input price: the harness does not rely on
|
|
49
|
+
# provider-side prompt caching.
|
|
50
|
+
"deepseek-flash": (0.30, 1.20),
|
|
51
|
+
"deepseek-v4-pro": (1.32, 3.96),
|
|
52
|
+
# Gemini paid tier, from ai.google.dev/gemini-api/docs/pricing on
|
|
53
|
+
# 2026-09-19. Output prices include thinking tokens.
|
|
54
|
+
"gemini-3.5-flash-lite": (0.30, 2.50),
|
|
55
|
+
"gemini-3.1-flash-lite": (0.25, 1.50),
|
|
56
|
+
"gemini-2.5-flash-lite": (0.10, 0.40),
|
|
57
|
+
# OpenAI list price (developers.openai.com/api/docs/pricing, 2026-09-19).
|
|
58
|
+
"gpt-5.6-luna": (0.20, 1.20),
|
|
59
|
+
# Through OpenRouter, from its public /api/v1/models listing on
|
|
60
|
+
# 2026-09-19. Reasoning tokens are billed as output.
|
|
61
|
+
"openai/gpt-5.6-luna": (0.20, 1.20),
|
|
62
|
+
"qwen/qwen3.8-flash": (0.15, 0.47),
|
|
63
|
+
"qwen/qwen3-235b-a22b-2507": (0.0875, 0.35),
|
|
64
|
+
"meta-llama/llama-4-maverick": (0.188, 0.652),
|
|
65
|
+
"meta-llama/llama-4-scout": (0.10, 0.30),
|
|
66
|
+
"meta-llama/llama-3.3-70b-instruct": (0.10, 0.32),
|
|
41
67
|
}
|
|
42
68
|
|
|
43
69
|
|
|
70
|
+
def _env_price() -> Optional[tuple]:
|
|
71
|
+
"""RHYLTHYME_EVAL_PRICE_IN / _OUT (USD per million tokens) price a model
|
|
72
|
+
the table does not know, e.g. one reached through an aggregator."""
|
|
73
|
+
try:
|
|
74
|
+
return (
|
|
75
|
+
float(os.environ["RHYLTHYME_EVAL_PRICE_IN"]),
|
|
76
|
+
float(os.environ["RHYLTHYME_EVAL_PRICE_OUT"]),
|
|
77
|
+
)
|
|
78
|
+
except (KeyError, ValueError):
|
|
79
|
+
return None
|
|
80
|
+
|
|
81
|
+
|
|
44
82
|
def price_for(model: str) -> Optional[tuple]:
|
|
45
83
|
"""``(input, output)`` USD per million tokens, or ``None`` if unknown."""
|
|
46
84
|
if model in PRICES_PER_MTOK:
|
|
@@ -49,7 +87,7 @@ def price_for(model: str) -> Optional[tuple]:
|
|
|
49
87
|
for key in PRICES_PER_MTOK:
|
|
50
88
|
if model.startswith(key) and (best is None or len(key) > len(best)):
|
|
51
89
|
best = key
|
|
52
|
-
return PRICES_PER_MTOK[best] if best else
|
|
90
|
+
return PRICES_PER_MTOK[best] if best else _env_price()
|
|
53
91
|
|
|
54
92
|
|
|
55
93
|
def estimate_cost(model: str, input_tokens: int, output_tokens: int) -> Optional[float]:
|
|
@@ -165,6 +203,143 @@ class AnthropicClient:
|
|
|
165
203
|
)
|
|
166
204
|
|
|
167
205
|
|
|
206
|
+
# Providers that speak the OpenAI chat-completions format. A model id picks
|
|
207
|
+
# its provider by prefix; an id with a slash ("deepseek/deepseek-flash") goes
|
|
208
|
+
# to OpenRouter, which fronts most of them under one key. RHYLTHYME_EVAL_BASE_URL
|
|
209
|
+
# and RHYLTHYME_EVAL_API_KEY override both, for any other compatible endpoint
|
|
210
|
+
# (a local vLLM or Ollama server included).
|
|
211
|
+
OPENAI_COMPAT_PROVIDERS = [
|
|
212
|
+
("deepseek-", "https://api.deepseek.com", "DEEPSEEK_API_KEY"),
|
|
213
|
+
(
|
|
214
|
+
"qwen",
|
|
215
|
+
"https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
|
|
216
|
+
"DASHSCOPE_API_KEY",
|
|
217
|
+
),
|
|
218
|
+
("kimi-", "https://api.moonshot.ai/v1", "MOONSHOT_API_KEY"),
|
|
219
|
+
("moonshot-", "https://api.moonshot.ai/v1", "MOONSHOT_API_KEY"),
|
|
220
|
+
("glm-", "https://api.z.ai/api/paas/v4", "ZAI_API_KEY"),
|
|
221
|
+
(
|
|
222
|
+
"gemini-",
|
|
223
|
+
"https://generativelanguage.googleapis.com/v1beta/openai",
|
|
224
|
+
"GEMINI_API_KEY",
|
|
225
|
+
),
|
|
226
|
+
("gpt-", "https://api.openai.com/v1", "OPENAI_API_KEY"),
|
|
227
|
+
]
|
|
228
|
+
OPENROUTER = ("https://openrouter.ai/api/v1", "OPENROUTER_API_KEY")
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def provider_for(model: str) -> Optional[tuple]:
|
|
232
|
+
"""``(base_url, api_key_env)`` for an OpenAI-compatible model, else None."""
|
|
233
|
+
if os.environ.get("RHYLTHYME_EVAL_BASE_URL"):
|
|
234
|
+
return (
|
|
235
|
+
os.environ["RHYLTHYME_EVAL_BASE_URL"].rstrip("/"),
|
|
236
|
+
"RHYLTHYME_EVAL_API_KEY",
|
|
237
|
+
)
|
|
238
|
+
if model.startswith("claude-"):
|
|
239
|
+
return None
|
|
240
|
+
if "/" in model:
|
|
241
|
+
return OPENROUTER
|
|
242
|
+
for prefix, base_url, key_env in OPENAI_COMPAT_PROVIDERS:
|
|
243
|
+
if model.startswith(prefix):
|
|
244
|
+
return (base_url, key_env)
|
|
245
|
+
return None
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
class OpenAICompatClient:
|
|
249
|
+
"""Chat-completions client for OpenAI-compatible endpoints. Standard
|
|
250
|
+
library only. Retries on 429/5xx; never on a timeout, because a request
|
|
251
|
+
that timed out here may still have been billed there."""
|
|
252
|
+
|
|
253
|
+
RETRY_STATUS = (429, 500, 502, 503, 529)
|
|
254
|
+
|
|
255
|
+
def __init__(
|
|
256
|
+
self,
|
|
257
|
+
base_url: str,
|
|
258
|
+
api_key: Optional[str],
|
|
259
|
+
*,
|
|
260
|
+
timeout: float = 900.0,
|
|
261
|
+
retries: int = 3,
|
|
262
|
+
):
|
|
263
|
+
self.base_url = base_url.rstrip("/")
|
|
264
|
+
self.api_key = api_key
|
|
265
|
+
# RHYLTHYME_EVAL_TIMEOUT (seconds) overrides the per-call timeout.
|
|
266
|
+
self.timeout = float(os.environ.get("RHYLTHYME_EVAL_TIMEOUT") or timeout)
|
|
267
|
+
self.retries = retries
|
|
268
|
+
|
|
269
|
+
def complete(
|
|
270
|
+
self,
|
|
271
|
+
messages: Sequence[Message],
|
|
272
|
+
*,
|
|
273
|
+
system: Optional[str] = None,
|
|
274
|
+
model: str,
|
|
275
|
+
max_tokens: int,
|
|
276
|
+
) -> Completion:
|
|
277
|
+
chat: List[Dict[str, Any]] = (
|
|
278
|
+
[{"role": "system", "content": system}] if system else []
|
|
279
|
+
)
|
|
280
|
+
chat += [{"role": m["role"], "content": m["content"]} for m in messages]
|
|
281
|
+
body = json.dumps(
|
|
282
|
+
{
|
|
283
|
+
"model": model,
|
|
284
|
+
"messages": chat,
|
|
285
|
+
# OpenAI's current models reject `max_tokens`.
|
|
286
|
+
(
|
|
287
|
+
"max_completion_tokens"
|
|
288
|
+
if "api.openai.com" in self.base_url
|
|
289
|
+
else "max_tokens"
|
|
290
|
+
): max_tokens,
|
|
291
|
+
"stream": False,
|
|
292
|
+
}
|
|
293
|
+
).encode("utf-8")
|
|
294
|
+
headers = {"Content-Type": "application/json"}
|
|
295
|
+
if self.api_key:
|
|
296
|
+
headers["Authorization"] = f"Bearer {self.api_key}"
|
|
297
|
+
for attempt in range(self.retries + 1):
|
|
298
|
+
req = urllib.request.Request(
|
|
299
|
+
f"{self.base_url}/chat/completions",
|
|
300
|
+
data=body,
|
|
301
|
+
headers=headers,
|
|
302
|
+
method="POST",
|
|
303
|
+
)
|
|
304
|
+
try:
|
|
305
|
+
with urllib.request.urlopen(req, timeout=self.timeout) as resp:
|
|
306
|
+
data = json.loads(resp.read().decode("utf-8"))
|
|
307
|
+
break
|
|
308
|
+
except urllib.error.HTTPError as exc:
|
|
309
|
+
detail = exc.read().decode("utf-8", "replace")[:300]
|
|
310
|
+
if exc.code in self.RETRY_STATUS and attempt < self.retries:
|
|
311
|
+
time.sleep(2.0 * (attempt + 1))
|
|
312
|
+
continue
|
|
313
|
+
raise RuntimeError(
|
|
314
|
+
f"{self.base_url} returned HTTP {exc.code}: {detail}"
|
|
315
|
+
) from exc
|
|
316
|
+
except (TimeoutError, urllib.error.URLError) as exc:
|
|
317
|
+
# A stalled provider. Retrying can double-bill a request that
|
|
318
|
+
# did complete remotely, so it is opt-in (cheap models only).
|
|
319
|
+
if (
|
|
320
|
+
os.environ.get("RHYLTHYME_EVAL_RETRY_TIMEOUTS")
|
|
321
|
+
and attempt < self.retries
|
|
322
|
+
):
|
|
323
|
+
continue
|
|
324
|
+
raise RuntimeError(
|
|
325
|
+
f"{self.base_url} did not answer within {self.timeout:.0f}s: {exc}"
|
|
326
|
+
) from exc
|
|
327
|
+
choice = (data.get("choices") or [{}])[0]
|
|
328
|
+
usage = data.get("usage") or {}
|
|
329
|
+
finish = choice.get("finish_reason")
|
|
330
|
+
return Completion(
|
|
331
|
+
text=str((choice.get("message") or {}).get("content") or ""),
|
|
332
|
+
input_tokens=int(usage.get("prompt_tokens") or 0),
|
|
333
|
+
# Reasoning tokens are billed as output and are inside this count.
|
|
334
|
+
output_tokens=int(usage.get("completion_tokens") or 0),
|
|
335
|
+
model=str(data.get("model") or model),
|
|
336
|
+
stop_reason={"stop": "end_turn", "length": "max_tokens"}.get(
|
|
337
|
+
finish, finish
|
|
338
|
+
),
|
|
339
|
+
raw=data,
|
|
340
|
+
)
|
|
341
|
+
|
|
342
|
+
|
|
168
343
|
CannedResponse = Union[str, Completion]
|
|
169
344
|
|
|
170
345
|
|
|
@@ -252,8 +427,20 @@ class FakeClient:
|
|
|
252
427
|
)
|
|
253
428
|
|
|
254
429
|
|
|
255
|
-
def make_client(
|
|
256
|
-
|
|
430
|
+
def make_client(
|
|
431
|
+
fake_responses: Optional[Any] = None, model: Optional[str] = None
|
|
432
|
+
) -> Client:
|
|
433
|
+
"""Build the client the CLI uses: a fake when canned responses are given,
|
|
434
|
+
otherwise the provider the model id belongs to (Anthropic by default)."""
|
|
257
435
|
if fake_responses is not None:
|
|
258
436
|
return FakeClient(fake_responses)
|
|
259
|
-
|
|
437
|
+
provider = provider_for(model or "")
|
|
438
|
+
if provider is None:
|
|
439
|
+
return AnthropicClient()
|
|
440
|
+
base_url, key_env = provider
|
|
441
|
+
api_key = os.environ.get(key_env)
|
|
442
|
+
if not api_key and "localhost" not in base_url and "127.0.0.1" not in base_url:
|
|
443
|
+
raise RuntimeError(
|
|
444
|
+
f"Model {model!r} is served by {base_url}; set {key_env} to use it."
|
|
445
|
+
)
|
|
446
|
+
return OpenAICompatClient(base_url, api_key)
|