rhylthyme-cli-runner 0.2.0a0__tar.gz → 0.2.2a0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. {rhylthyme_cli_runner-0.2.0a0/src/rhylthyme_cli_runner.egg-info → rhylthyme_cli_runner-0.2.2a0}/PKG-INFO +154 -3
  2. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/README.md +153 -2
  3. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/setup.py +1 -1
  4. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/__init__.py +1 -1
  5. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/cli.py +4 -1
  6. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/llm.py +191 -4
  7. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/program_planner.py +20 -4
  8. rhylthyme_cli_runner-0.2.2a0/src/rhylthyme_cli_runner/remote/checks.py +663 -0
  9. rhylthyme_cli_runner-0.2.2a0/src/rhylthyme_cli_runner/remote/cli.py +594 -0
  10. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/remote/mcp_client.py +69 -16
  11. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0/src/rhylthyme_cli_runner.egg-info}/PKG-INFO +154 -3
  12. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner.egg-info/SOURCES.txt +5 -0
  13. rhylthyme_cli_runner-0.2.2a0/tests/test_eval_providers.py +292 -0
  14. rhylthyme_cli_runner-0.2.2a0/tests/test_mcp_checks.py +441 -0
  15. rhylthyme_cli_runner-0.2.2a0/tests/test_planner_time_strings.py +20 -0
  16. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_remote.py +95 -0
  17. rhylthyme_cli_runner-0.2.2a0/tests/test_skill.py +71 -0
  18. rhylthyme_cli_runner-0.2.0a0/src/rhylthyme_cli_runner/remote/cli.py +0 -282
  19. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/LICENSE +0 -0
  20. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/pyproject.toml +0 -0
  21. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/setup.cfg +0 -0
  22. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/cli.py +0 -0
  23. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/environment_icons.py +0 -0
  24. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/environment_loader.py +0 -0
  25. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/environment_schemas.py +0 -0
  26. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/__init__.py +0 -0
  27. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/compare.py +0 -0
  28. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/gold.py +0 -0
  29. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/harness.py +0 -0
  30. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/matcher.py +0 -0
  31. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/metrics.py +0 -0
  32. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/__init__.py +0 -0
  33. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/authoring_guide.md +0 -0
  34. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/baseline.md +0 -0
  35. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/baseline.py +0 -0
  36. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/extract.py +0 -0
  37. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/patterns/four_turn.py +0 -0
  38. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/eval/report.py +0 -0
  39. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/expand_replicates.py +0 -0
  40. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/__init__.py +0 -0
  41. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/calibrate.py +0 -0
  42. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/calibrate_cli.py +0 -0
  43. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/cli.py +0 -0
  44. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/evaluate.py +0 -0
  45. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/evaluate_cli.py +0 -0
  46. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/factors.py +0 -0
  47. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/hash.py +0 -0
  48. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/predict.py +0 -0
  49. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/recorder.py +0 -0
  50. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/render.py +0 -0
  51. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/replay.py +0 -0
  52. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/report.py +0 -0
  53. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/report_cli.py +0 -0
  54. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/store.py +0 -0
  55. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/synth.py +0 -0
  56. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/history/usable.py +0 -0
  57. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/instance_checks.py +0 -0
  58. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/program_runner.py +0 -0
  59. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/remote/__init__.py +0 -0
  60. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/remote/auth.py +0 -0
  61. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner/validate_program.py +0 -0
  62. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner.egg-info/dependency_links.txt +0 -0
  63. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner.egg-info/entry_points.txt +0 -0
  64. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner.egg-info/requires.txt +0 -0
  65. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/src/rhylthyme_cli_runner.egg-info/top_level.txt +0 -0
  66. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_calibrate.py +0 -0
  67. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_cli.py +0 -0
  68. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_eval_baseline.py +0 -0
  69. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_eval_harness.py +0 -0
  70. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_eval_scorer.py +0 -0
  71. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_evaluate.py +0 -0
  72. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_example_validation.py +0 -0
  73. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_examples_integration.py +0 -0
  74. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_expand_replicates.py +0 -0
  75. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_hash_parity.py +0 -0
  76. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_history_report.py +0 -0
  77. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_predict.py +0 -0
  78. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_program_runner.py +0 -0
  79. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_replay.py +0 -0
  80. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_run_recorder.py +0 -0
  81. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_synth.py +0 -0
  82. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_usable.py +0 -0
  83. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_validate_examples_ci.py +0 -0
  84. {rhylthyme_cli_runner-0.2.0a0 → rhylthyme_cli_runner-0.2.2a0}/tests/test_validate_program.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rhylthyme-cli-runner
3
- Version: 0.2.0a0
3
+ Version: 0.2.2a0
4
4
  Summary: CLI runner for Rhylthyme real-time program schedules
5
5
  Home-page: https://github.com/rhylthyme/rhylthyme-cli-runner
6
6
  Author: Rhylthyme Team
@@ -197,9 +197,30 @@ rhylthyme run examples/programs/breakfast_schedule.json --environment kitchen
197
197
  rhylthyme run examples/programs/breakfast_schedule.json --no-validate
198
198
  ```
199
199
 
200
+ ### Analyze and Publish a Program
201
+
202
+ Neither needs an account; both are computed by the hosted MCP server.
203
+
204
+ ```bash
205
+ # Total length, critical path, resource conflicts
206
+ rhylthyme analyze examples/programs/breakfast_schedule.json
207
+
208
+ # When does each step start if breakfast is at 8:30?
209
+ rhylthyme analyze examples/programs/breakfast_schedule.json --finish-at 8:30am
210
+
211
+ # A live, shareable timeline with timers; prints the URL
212
+ rhylthyme publish examples/programs/breakfast_schedule.json
213
+ ```
214
+
215
+ `validate` checks structure only. Two steps that want the only oven at the
216
+ same time pass validation; `analyze` reports them, and `--strict` makes that
217
+ a non-zero exit.
218
+
200
219
  ### Optimize a Program
201
220
 
202
- Create an optimized version of a program to reduce resource contention:
221
+ `plan` is an older stagger heuristic that reads the pre-0.2 program format; on
222
+ programs written with `stepId` and `task` (all current examples) it writes the
223
+ program back unchanged. Use `analyze` to find contention.
203
224
 
204
225
  ```bash
205
226
  # Optimize a program and save to new file
@@ -227,6 +248,29 @@ rhylthyme validate-environments
227
248
  rhylthyme environment-info kitchen
228
249
  ```
229
250
 
251
+ ## Claude Skill
252
+
253
+ [`skills/rhylthyme`](skills/rhylthyme) is an [Agent Skill](https://docs.claude.com/en/docs/agents-and-tools/agent-skills/overview)
254
+ that teaches Claude (Claude Code, the Claude apps, the Agent SDK) to turn a
255
+ protocol, recipe or run sheet into a validated program with this CLI, check
256
+ it for conflicts, and hand back a live timeline. It follows the layout of
257
+ [K-Dense scientific skills](https://github.com/K-Dense-AI/claude-scientific-skills):
258
+ a `SKILL.md` plus `references/`.
259
+
260
+ ```bash
261
+ # Claude Code, for one user
262
+ mkdir -p ~/.claude/skills
263
+ cp -r skills/rhylthyme ~/.claude/skills/
264
+
265
+ # or for one project
266
+ mkdir -p .claude/skills && cp -r skills/rhylthyme .claude/skills/
267
+ ```
268
+
269
+ Then ask, for example, "time a Western blot so imaging is at 4 pm" or "two PCR
270
+ protocols, one thermocycler: when do I start each?". `tests/test_skill.py`
271
+ validates every program in the skill and checks that every command it names
272
+ exists.
273
+
230
274
  ## Program File Examples
231
275
 
232
276
  ### Simple Breakfast Schedule
@@ -305,6 +349,39 @@ Environment: `RHYLTHYME_TOKEN` (access token, overrides the stored session),
305
349
  `RHYLTHYME_MCP_URL` (default `https://mcp.rhylthyme.com/mcp`),
306
350
  `RHYLTHYME_SITE_URL` (default `https://www.rhylthyme.com`).
307
351
 
352
+ ### `rhylthyme mcp-test`
353
+
354
+ Smoke-tests a Rhylthyme MCP server and exits 1 if any check fails. By
355
+ default it runs the read-only checks against all five hosted endpoints
356
+ (`/mcp`, `/kitchen/mcp`, `/lab/mcp`, `/events/mcp`, `/gym/mcp`):
357
+
358
+ | Check | What it proves |
359
+ |---|---|
360
+ | `initialize` | server name, protocol version, tools/resources/prompts capabilities, instructions |
361
+ | `tools` | core tools and the endpoint's one-shot tools are listed, each with a description and schema |
362
+ | `validate-good` / `validate-bad` | a valid program passes (including a `type: "compound"` trigger); a dangling step reference is rejected with a fix hint |
363
+ | `analyze` | makespan, critical path and wall-clock itinerary for a known program |
364
+ | `resources` / `prompts` | schema and guides are readable, a bundled example validates, `plan_schedule` substitutes its arguments |
365
+ | `json-accept` | clients that do not accept SSE (`*/*`, `application/json`) get JSON, not 406 |
366
+ | `bad-requests` | unknown method gives -32601, unknown tool gives an error, never a 5xx |
367
+ | `login-gate` | account tools refuse without a token and point at `login` |
368
+ | `catalog` | public search returns entries that load with a URL (empty catalog = warning) |
369
+ | `publish` (`--publish`) | `visualize_schedule` returns a URL on the right site; the page and PNG load |
370
+ | `generate` (`--generate`) | the model-backed `import_text` returns a program that validates (needs `login`) |
371
+
372
+ ```bash
373
+ rhylthyme mcp-test # everything read-only
374
+ rhylthyme mcp-test -e lab --publish # one endpoint, plus a real share
375
+ rhylthyme mcp-test -k catalog -k tools # only some checks
376
+ rhylthyme mcp-test --url http://localhost:3000/mcp -e generic
377
+ rhylthyme mcp-test --json --strict # cron / CI: warnings fail too
378
+ ```
379
+
380
+ Requests carry `mcp-test` in the User-Agent so the hosted server logs the
381
+ errors these checks provoke without alerting anyone. The same suite is
382
+ importable (`rhylthyme_cli_runner.remote.checks.run_suite`), and
383
+ `RHYLTHYME_MCP_LIVE=1 pytest tests/test_mcp_checks.py` runs it live.
384
+
308
385
  ### `rhylthyme login` / `logout` / `whoami`
309
386
 
310
387
  `login` opens the rhylthyme.com sign-in page and receives the session on a
@@ -331,9 +408,33 @@ Runs programs with interactive terminal UI.
331
408
  - `--validate / --no-validate`: Validate before running (default: True)
332
409
  - `--auto-start`: Automatically start without manual trigger
333
410
 
411
+ ### `rhylthyme analyze`
412
+
413
+ Total length, critical path, what gates each link of it, resource conflicts
414
+ and tracks that finish early. No sign-in; nothing is published.
415
+
416
+ **Options:**
417
+ - `--finish-at TEXT`: when everything must be finished (`19:00`, `7:30pm` or ISO 8601); prints local clock times per step
418
+ - `--start-at TEXT`: when the program starts; ignored with `--finish-at`
419
+ - `--strict`: exit non-zero when there are resource conflicts
420
+ - `--json`: the full analysis
421
+
422
+ ### `rhylthyme publish`
423
+
424
+ Publishes a program file as a live timeline and prints its URL. No sign-in.
425
+ A published timeline is reachable by anyone who has the link.
426
+
427
+ **Options:**
428
+ - `-e, --env`: `generic`, `kitchen`, `lab`, `events`, `gym` (default: from `environmentType`)
429
+ - `-q, --quiet`: print only the URL
430
+ - `--json`: `url`, `shareId`, `imageUrl`, `makespanSeconds`, `warnings`
431
+ - `--open`: open it in a browser
432
+
334
433
  ### `rhylthyme plan`
335
434
 
336
- Optimizes program schedules to reduce resource contention.
435
+ An older stagger heuristic. It reads the pre-0.2 program format and leaves
436
+ programs written with `stepId` and `task` unchanged; use `analyze` to find
437
+ contention and edit the triggers.
337
438
 
338
439
  **Options:**
339
440
  - `--verbose, -v`: Show detailed planning information
@@ -614,6 +715,56 @@ make test-unit # excludes llm
614
715
  RHYLTHYME_EVAL_LIVE=1 pytest -m llm # one real call, opt in
615
716
  ```
616
717
 
718
+
719
+ ### Other models and a spending cap
720
+
721
+ `--model` also accepts models served through an OpenAI-compatible API. The
722
+ provider is picked from the model id and the key is read from the
723
+ environment:
724
+
725
+ | Model id | Provider | Key |
726
+ |---|---|---|
727
+ | `claude-*` | Anthropic | `ANTHROPIC_API_KEY` |
728
+ | `deepseek-*` (e.g. `deepseek-flash`, `deepseek-v4-pro`) | api.deepseek.com | `DEEPSEEK_API_KEY` |
729
+ | `qwen*` | Alibaba DashScope (international) | `DASHSCOPE_API_KEY` |
730
+ | `kimi-*`, `moonshot-*` | Moonshot | `MOONSHOT_API_KEY` |
731
+ | `glm-*` | Z.ai | `ZAI_API_KEY` |
732
+ | anything with a slash, e.g. `deepseek/deepseek-flash` | OpenRouter | `OPENROUTER_API_KEY` |
733
+ | any id, with `RHYLTHYME_EVAL_BASE_URL` set | that endpoint (vLLM, Ollama, ...) | `RHYLTHYME_EVAL_API_KEY`, optional for localhost |
734
+
735
+ Costs come from the price table in `eval/llm.py` (DeepSeek at its peak
736
+ rates, so estimates are upper bounds). For a model the table does not know,
737
+ set `RHYLTHYME_EVAL_PRICE_IN` and `RHYLTHYME_EVAL_PRICE_OUT` in USD per
738
+ million tokens.
739
+
740
+ `eval/run_model_comparison.py` runs a model under a hard budget. It runs one
741
+ gold program at a time, both prompts per program, records each program's
742
+ cost in `eval/models/spend-ledger.json`, and stops before the next program
743
+ could cross `--cap`. The ledger is cumulative across models and
744
+ invocations, and the script refuses a model with no known price (it would
745
+ be recorded as $0 and the cap would never trip).
746
+
747
+ One OpenRouter key reaches OpenAI, Qwen and Meta models (ids with a vendor
748
+ prefix). Run the cheapest first and raise the cumulative cap as you go, so
749
+ no single model can spend the whole allowance. Reasoning models need a
750
+ higher per-call output ceiling than the default 16,000 tokens:
751
+
752
+ ```bash
753
+ export OPENROUTER_API_KEY=...
754
+ L=eval/models/spend-ledger-openrouter.json
755
+ python eval/run_model_comparison.py --ledger $L --model meta-llama/llama-4-maverick --cap 0.80 --first-guess 0.05
756
+ python eval/run_model_comparison.py --ledger $L --model qwen/qwen3.8-flash --cap 2.00 --first-guess 0.08 --max-tokens 64000
757
+ python eval/run_model_comparison.py --ledger $L --model openai/gpt-5.6-luna --cap 4.50 --first-guess 0.15 --max-tokens 64000
758
+ python eval/compare_models.py
759
+ ```
760
+
761
+ ```bash
762
+ export DEEPSEEK_API_KEY=...
763
+ python eval/run_model_comparison.py --model deepseek-flash --cap 6.70 --dry-run
764
+ python eval/run_model_comparison.py --model deepseek-flash --cap 6.70
765
+ python eval/compare_models.py # like-for-like table across every model run so far
766
+ ```
767
+
617
768
  ## Development
618
769
 
619
770
  1. Clone the repository
@@ -137,9 +137,30 @@ rhylthyme run examples/programs/breakfast_schedule.json --environment kitchen
137
137
  rhylthyme run examples/programs/breakfast_schedule.json --no-validate
138
138
  ```
139
139
 
140
+ ### Analyze and Publish a Program
141
+
142
+ Neither needs an account; both are computed by the hosted MCP server.
143
+
144
+ ```bash
145
+ # Total length, critical path, resource conflicts
146
+ rhylthyme analyze examples/programs/breakfast_schedule.json
147
+
148
+ # When does each step start if breakfast is at 8:30?
149
+ rhylthyme analyze examples/programs/breakfast_schedule.json --finish-at 8:30am
150
+
151
+ # A live, shareable timeline with timers; prints the URL
152
+ rhylthyme publish examples/programs/breakfast_schedule.json
153
+ ```
154
+
155
+ `validate` checks structure only. Two steps that want the only oven at the
156
+ same time pass validation; `analyze` reports them, and `--strict` makes that
157
+ a non-zero exit.
158
+
140
159
  ### Optimize a Program
141
160
 
142
- Create an optimized version of a program to reduce resource contention:
161
+ `plan` is an older stagger heuristic that reads the pre-0.2 program format; on
162
+ programs written with `stepId` and `task` (all current examples) it writes the
163
+ program back unchanged. Use `analyze` to find contention.
143
164
 
144
165
  ```bash
145
166
  # Optimize a program and save to new file
@@ -167,6 +188,29 @@ rhylthyme validate-environments
167
188
  rhylthyme environment-info kitchen
168
189
  ```
169
190
 
191
+ ## Claude Skill
192
+
193
+ [`skills/rhylthyme`](skills/rhylthyme) is an [Agent Skill](https://docs.claude.com/en/docs/agents-and-tools/agent-skills/overview)
194
+ that teaches Claude (Claude Code, the Claude apps, the Agent SDK) to turn a
195
+ protocol, recipe or run sheet into a validated program with this CLI, check
196
+ it for conflicts, and hand back a live timeline. It follows the layout of
197
+ [K-Dense scientific skills](https://github.com/K-Dense-AI/claude-scientific-skills):
198
+ a `SKILL.md` plus `references/`.
199
+
200
+ ```bash
201
+ # Claude Code, for one user
202
+ mkdir -p ~/.claude/skills
203
+ cp -r skills/rhylthyme ~/.claude/skills/
204
+
205
+ # or for one project
206
+ mkdir -p .claude/skills && cp -r skills/rhylthyme .claude/skills/
207
+ ```
208
+
209
+ Then ask, for example, "time a Western blot so imaging is at 4 pm" or "two PCR
210
+ protocols, one thermocycler: when do I start each?". `tests/test_skill.py`
211
+ validates every program in the skill and checks that every command it names
212
+ exists.
213
+
170
214
  ## Program File Examples
171
215
 
172
216
  ### Simple Breakfast Schedule
@@ -245,6 +289,39 @@ Environment: `RHYLTHYME_TOKEN` (access token, overrides the stored session),
245
289
  `RHYLTHYME_MCP_URL` (default `https://mcp.rhylthyme.com/mcp`),
246
290
  `RHYLTHYME_SITE_URL` (default `https://www.rhylthyme.com`).
247
291
 
292
+ ### `rhylthyme mcp-test`
293
+
294
+ Smoke-tests a Rhylthyme MCP server and exits 1 if any check fails. By
295
+ default it runs the read-only checks against all five hosted endpoints
296
+ (`/mcp`, `/kitchen/mcp`, `/lab/mcp`, `/events/mcp`, `/gym/mcp`):
297
+
298
+ | Check | What it proves |
299
+ |---|---|
300
+ | `initialize` | server name, protocol version, tools/resources/prompts capabilities, instructions |
301
+ | `tools` | core tools and the endpoint's one-shot tools are listed, each with a description and schema |
302
+ | `validate-good` / `validate-bad` | a valid program passes (including a `type: "compound"` trigger); a dangling step reference is rejected with a fix hint |
303
+ | `analyze` | makespan, critical path and wall-clock itinerary for a known program |
304
+ | `resources` / `prompts` | schema and guides are readable, a bundled example validates, `plan_schedule` substitutes its arguments |
305
+ | `json-accept` | clients that do not accept SSE (`*/*`, `application/json`) get JSON, not 406 |
306
+ | `bad-requests` | unknown method gives -32601, unknown tool gives an error, never a 5xx |
307
+ | `login-gate` | account tools refuse without a token and point at `login` |
308
+ | `catalog` | public search returns entries that load with a URL (empty catalog = warning) |
309
+ | `publish` (`--publish`) | `visualize_schedule` returns a URL on the right site; the page and PNG load |
310
+ | `generate` (`--generate`) | the model-backed `import_text` returns a program that validates (needs `login`) |
311
+
312
+ ```bash
313
+ rhylthyme mcp-test # everything read-only
314
+ rhylthyme mcp-test -e lab --publish # one endpoint, plus a real share
315
+ rhylthyme mcp-test -k catalog -k tools # only some checks
316
+ rhylthyme mcp-test --url http://localhost:3000/mcp -e generic
317
+ rhylthyme mcp-test --json --strict # cron / CI: warnings fail too
318
+ ```
319
+
320
+ Requests carry `mcp-test` in the User-Agent so the hosted server logs the
321
+ errors these checks provoke without alerting anyone. The same suite is
322
+ importable (`rhylthyme_cli_runner.remote.checks.run_suite`), and
323
+ `RHYLTHYME_MCP_LIVE=1 pytest tests/test_mcp_checks.py` runs it live.
324
+
248
325
  ### `rhylthyme login` / `logout` / `whoami`
249
326
 
250
327
  `login` opens the rhylthyme.com sign-in page and receives the session on a
@@ -271,9 +348,33 @@ Runs programs with interactive terminal UI.
271
348
  - `--validate / --no-validate`: Validate before running (default: True)
272
349
  - `--auto-start`: Automatically start without manual trigger
273
350
 
351
+ ### `rhylthyme analyze`
352
+
353
+ Total length, critical path, what gates each link of it, resource conflicts
354
+ and tracks that finish early. No sign-in; nothing is published.
355
+
356
+ **Options:**
357
+ - `--finish-at TEXT`: when everything must be finished (`19:00`, `7:30pm` or ISO 8601); prints local clock times per step
358
+ - `--start-at TEXT`: when the program starts; ignored with `--finish-at`
359
+ - `--strict`: exit non-zero when there are resource conflicts
360
+ - `--json`: the full analysis
361
+
362
+ ### `rhylthyme publish`
363
+
364
+ Publishes a program file as a live timeline and prints its URL. No sign-in.
365
+ A published timeline is reachable by anyone who has the link.
366
+
367
+ **Options:**
368
+ - `-e, --env`: `generic`, `kitchen`, `lab`, `events`, `gym` (default: from `environmentType`)
369
+ - `-q, --quiet`: print only the URL
370
+ - `--json`: `url`, `shareId`, `imageUrl`, `makespanSeconds`, `warnings`
371
+ - `--open`: open it in a browser
372
+
274
373
  ### `rhylthyme plan`
275
374
 
276
- Optimizes program schedules to reduce resource contention.
375
+ An older stagger heuristic. It reads the pre-0.2 program format and leaves
376
+ programs written with `stepId` and `task` unchanged; use `analyze` to find
377
+ contention and edit the triggers.
277
378
 
278
379
  **Options:**
279
380
  - `--verbose, -v`: Show detailed planning information
@@ -554,6 +655,56 @@ make test-unit # excludes llm
554
655
  RHYLTHYME_EVAL_LIVE=1 pytest -m llm # one real call, opt in
555
656
  ```
556
657
 
658
+
659
+ ### Other models and a spending cap
660
+
661
+ `--model` also accepts models served through an OpenAI-compatible API. The
662
+ provider is picked from the model id and the key is read from the
663
+ environment:
664
+
665
+ | Model id | Provider | Key |
666
+ |---|---|---|
667
+ | `claude-*` | Anthropic | `ANTHROPIC_API_KEY` |
668
+ | `deepseek-*` (e.g. `deepseek-flash`, `deepseek-v4-pro`) | api.deepseek.com | `DEEPSEEK_API_KEY` |
669
+ | `qwen*` | Alibaba DashScope (international) | `DASHSCOPE_API_KEY` |
670
+ | `kimi-*`, `moonshot-*` | Moonshot | `MOONSHOT_API_KEY` |
671
+ | `glm-*` | Z.ai | `ZAI_API_KEY` |
672
+ | anything with a slash, e.g. `deepseek/deepseek-flash` | OpenRouter | `OPENROUTER_API_KEY` |
673
+ | any id, with `RHYLTHYME_EVAL_BASE_URL` set | that endpoint (vLLM, Ollama, ...) | `RHYLTHYME_EVAL_API_KEY`, optional for localhost |
674
+
675
+ Costs come from the price table in `eval/llm.py` (DeepSeek at its peak
676
+ rates, so estimates are upper bounds). For a model the table does not know,
677
+ set `RHYLTHYME_EVAL_PRICE_IN` and `RHYLTHYME_EVAL_PRICE_OUT` in USD per
678
+ million tokens.
679
+
680
+ `eval/run_model_comparison.py` runs a model under a hard budget. It runs one
681
+ gold program at a time, both prompts per program, records each program's
682
+ cost in `eval/models/spend-ledger.json`, and stops before the next program
683
+ could cross `--cap`. The ledger is cumulative across models and
684
+ invocations, and the script refuses a model with no known price (it would
685
+ be recorded as $0 and the cap would never trip).
686
+
687
+ One OpenRouter key reaches OpenAI, Qwen and Meta models (ids with a vendor
688
+ prefix). Run the cheapest first and raise the cumulative cap as you go, so
689
+ no single model can spend the whole allowance. Reasoning models need a
690
+ higher per-call output ceiling than the default 16,000 tokens:
691
+
692
+ ```bash
693
+ export OPENROUTER_API_KEY=...
694
+ L=eval/models/spend-ledger-openrouter.json
695
+ python eval/run_model_comparison.py --ledger $L --model meta-llama/llama-4-maverick --cap 0.80 --first-guess 0.05
696
+ python eval/run_model_comparison.py --ledger $L --model qwen/qwen3.8-flash --cap 2.00 --first-guess 0.08 --max-tokens 64000
697
+ python eval/run_model_comparison.py --ledger $L --model openai/gpt-5.6-luna --cap 4.50 --first-guess 0.15 --max-tokens 64000
698
+ python eval/compare_models.py
699
+ ```
700
+
701
+ ```bash
702
+ export DEEPSEEK_API_KEY=...
703
+ python eval/run_model_comparison.py --model deepseek-flash --cap 6.70 --dry-run
704
+ python eval/run_model_comparison.py --model deepseek-flash --cap 6.70
705
+ python eval/compare_models.py # like-for-like table across every model run so far
706
+ ```
707
+
557
708
  ## Development
558
709
 
559
710
  1. Clone the repository
@@ -19,7 +19,7 @@ def read_readme():
19
19
 
20
20
  setup(
21
21
  name="rhylthyme-cli-runner",
22
- version="0.2.0a0",
22
+ version="0.2.2a0",
23
23
  description="CLI runner for Rhylthyme real-time program schedules",
24
24
  long_description=read_readme(),
25
25
  long_description_content_type="text/markdown",
@@ -6,7 +6,7 @@ This package provides the command-line interface for running and validating
6
6
  Rhylthyme real-time program schedules.
7
7
  """
8
8
 
9
- __version__ = "0.2.0a0"
9
+ __version__ = "0.2.2a0"
10
10
  __author__ = "Rhylthyme Team"
11
11
  __description__ = "CLI runner for Rhylthyme real-time program schedules"
12
12
 
@@ -321,7 +321,10 @@ def _live(
321
321
  if fake_client_path:
322
322
  with open(fake_client_path, "r", encoding="utf-8") as handle:
323
323
  fake_responses = json.load(handle)
324
- client = make_client(fake_responses)
324
+ try:
325
+ client = make_client(fake_responses, model=model)
326
+ except RuntimeError as exc:
327
+ raise click.UsageError(str(exc))
325
328
 
326
329
  gold_set = _load_gold(gold_dir)
327
330
  out_root = Path(out_dir)
@@ -16,7 +16,12 @@ Plus a small price table so runs can log an estimated cost.
16
16
 
17
17
  from __future__ import annotations
18
18
 
19
+ import json
20
+ import os
19
21
  import threading
22
+ import time
23
+ import urllib.error
24
+ import urllib.request
20
25
  from dataclasses import dataclass, field
21
26
  from typing import Any, Dict, List, Mapping, Optional, Protocol, Sequence, Union
22
27
 
@@ -38,9 +43,42 @@ PRICES_PER_MTOK: Dict[str, tuple] = {
38
43
  "claude-sonnet-5": (2.00, 10.00),
39
44
  "claude-sonnet-4-6": (3.00, 15.00),
40
45
  "claude-haiku-4-5": (1.00, 5.00),
46
+ # DeepSeek, from api-docs.deepseek.com/quick_start/pricing on 2026-09-19.
47
+ # PEAK rates (off-peak is half), so a budget computed from these is an
48
+ # upper bound. Cache-miss input price: the harness does not rely on
49
+ # provider-side prompt caching.
50
+ "deepseek-flash": (0.30, 1.20),
51
+ "deepseek-v4-pro": (1.32, 3.96),
52
+ # Gemini paid tier, from ai.google.dev/gemini-api/docs/pricing on
53
+ # 2026-09-19. Output prices include thinking tokens.
54
+ "gemini-3.5-flash-lite": (0.30, 2.50),
55
+ "gemini-3.1-flash-lite": (0.25, 1.50),
56
+ "gemini-2.5-flash-lite": (0.10, 0.40),
57
+ # OpenAI list price (developers.openai.com/api/docs/pricing, 2026-09-19).
58
+ "gpt-5.6-luna": (0.20, 1.20),
59
+ # Through OpenRouter, from its public /api/v1/models listing on
60
+ # 2026-09-19. Reasoning tokens are billed as output.
61
+ "openai/gpt-5.6-luna": (0.20, 1.20),
62
+ "qwen/qwen3.8-flash": (0.15, 0.47),
63
+ "qwen/qwen3-235b-a22b-2507": (0.0875, 0.35),
64
+ "meta-llama/llama-4-maverick": (0.188, 0.652),
65
+ "meta-llama/llama-4-scout": (0.10, 0.30),
66
+ "meta-llama/llama-3.3-70b-instruct": (0.10, 0.32),
41
67
  }
42
68
 
43
69
 
70
+ def _env_price() -> Optional[tuple]:
71
+ """RHYLTHYME_EVAL_PRICE_IN / _OUT (USD per million tokens) price a model
72
+ the table does not know, e.g. one reached through an aggregator."""
73
+ try:
74
+ return (
75
+ float(os.environ["RHYLTHYME_EVAL_PRICE_IN"]),
76
+ float(os.environ["RHYLTHYME_EVAL_PRICE_OUT"]),
77
+ )
78
+ except (KeyError, ValueError):
79
+ return None
80
+
81
+
44
82
  def price_for(model: str) -> Optional[tuple]:
45
83
  """``(input, output)`` USD per million tokens, or ``None`` if unknown."""
46
84
  if model in PRICES_PER_MTOK:
@@ -49,7 +87,7 @@ def price_for(model: str) -> Optional[tuple]:
49
87
  for key in PRICES_PER_MTOK:
50
88
  if model.startswith(key) and (best is None or len(key) > len(best)):
51
89
  best = key
52
- return PRICES_PER_MTOK[best] if best else None
90
+ return PRICES_PER_MTOK[best] if best else _env_price()
53
91
 
54
92
 
55
93
  def estimate_cost(model: str, input_tokens: int, output_tokens: int) -> Optional[float]:
@@ -165,6 +203,143 @@ class AnthropicClient:
165
203
  )
166
204
 
167
205
 
206
+ # Providers that speak the OpenAI chat-completions format. A model id picks
207
+ # its provider by prefix; an id with a slash ("deepseek/deepseek-flash") goes
208
+ # to OpenRouter, which fronts most of them under one key. RHYLTHYME_EVAL_BASE_URL
209
+ # and RHYLTHYME_EVAL_API_KEY override both, for any other compatible endpoint
210
+ # (a local vLLM or Ollama server included).
211
+ OPENAI_COMPAT_PROVIDERS = [
212
+ ("deepseek-", "https://api.deepseek.com", "DEEPSEEK_API_KEY"),
213
+ (
214
+ "qwen",
215
+ "https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
216
+ "DASHSCOPE_API_KEY",
217
+ ),
218
+ ("kimi-", "https://api.moonshot.ai/v1", "MOONSHOT_API_KEY"),
219
+ ("moonshot-", "https://api.moonshot.ai/v1", "MOONSHOT_API_KEY"),
220
+ ("glm-", "https://api.z.ai/api/paas/v4", "ZAI_API_KEY"),
221
+ (
222
+ "gemini-",
223
+ "https://generativelanguage.googleapis.com/v1beta/openai",
224
+ "GEMINI_API_KEY",
225
+ ),
226
+ ("gpt-", "https://api.openai.com/v1", "OPENAI_API_KEY"),
227
+ ]
228
+ OPENROUTER = ("https://openrouter.ai/api/v1", "OPENROUTER_API_KEY")
229
+
230
+
231
+ def provider_for(model: str) -> Optional[tuple]:
232
+ """``(base_url, api_key_env)`` for an OpenAI-compatible model, else None."""
233
+ if os.environ.get("RHYLTHYME_EVAL_BASE_URL"):
234
+ return (
235
+ os.environ["RHYLTHYME_EVAL_BASE_URL"].rstrip("/"),
236
+ "RHYLTHYME_EVAL_API_KEY",
237
+ )
238
+ if model.startswith("claude-"):
239
+ return None
240
+ if "/" in model:
241
+ return OPENROUTER
242
+ for prefix, base_url, key_env in OPENAI_COMPAT_PROVIDERS:
243
+ if model.startswith(prefix):
244
+ return (base_url, key_env)
245
+ return None
246
+
247
+
248
+ class OpenAICompatClient:
249
+ """Chat-completions client for OpenAI-compatible endpoints. Standard
250
+ library only. Retries on 429/5xx; never on a timeout, because a request
251
+ that timed out here may still have been billed there."""
252
+
253
+ RETRY_STATUS = (429, 500, 502, 503, 529)
254
+
255
+ def __init__(
256
+ self,
257
+ base_url: str,
258
+ api_key: Optional[str],
259
+ *,
260
+ timeout: float = 900.0,
261
+ retries: int = 3,
262
+ ):
263
+ self.base_url = base_url.rstrip("/")
264
+ self.api_key = api_key
265
+ # RHYLTHYME_EVAL_TIMEOUT (seconds) overrides the per-call timeout.
266
+ self.timeout = float(os.environ.get("RHYLTHYME_EVAL_TIMEOUT") or timeout)
267
+ self.retries = retries
268
+
269
+ def complete(
270
+ self,
271
+ messages: Sequence[Message],
272
+ *,
273
+ system: Optional[str] = None,
274
+ model: str,
275
+ max_tokens: int,
276
+ ) -> Completion:
277
+ chat: List[Dict[str, Any]] = (
278
+ [{"role": "system", "content": system}] if system else []
279
+ )
280
+ chat += [{"role": m["role"], "content": m["content"]} for m in messages]
281
+ body = json.dumps(
282
+ {
283
+ "model": model,
284
+ "messages": chat,
285
+ # OpenAI's current models reject `max_tokens`.
286
+ (
287
+ "max_completion_tokens"
288
+ if "api.openai.com" in self.base_url
289
+ else "max_tokens"
290
+ ): max_tokens,
291
+ "stream": False,
292
+ }
293
+ ).encode("utf-8")
294
+ headers = {"Content-Type": "application/json"}
295
+ if self.api_key:
296
+ headers["Authorization"] = f"Bearer {self.api_key}"
297
+ for attempt in range(self.retries + 1):
298
+ req = urllib.request.Request(
299
+ f"{self.base_url}/chat/completions",
300
+ data=body,
301
+ headers=headers,
302
+ method="POST",
303
+ )
304
+ try:
305
+ with urllib.request.urlopen(req, timeout=self.timeout) as resp:
306
+ data = json.loads(resp.read().decode("utf-8"))
307
+ break
308
+ except urllib.error.HTTPError as exc:
309
+ detail = exc.read().decode("utf-8", "replace")[:300]
310
+ if exc.code in self.RETRY_STATUS and attempt < self.retries:
311
+ time.sleep(2.0 * (attempt + 1))
312
+ continue
313
+ raise RuntimeError(
314
+ f"{self.base_url} returned HTTP {exc.code}: {detail}"
315
+ ) from exc
316
+ except (TimeoutError, urllib.error.URLError) as exc:
317
+ # A stalled provider. Retrying can double-bill a request that
318
+ # did complete remotely, so it is opt-in (cheap models only).
319
+ if (
320
+ os.environ.get("RHYLTHYME_EVAL_RETRY_TIMEOUTS")
321
+ and attempt < self.retries
322
+ ):
323
+ continue
324
+ raise RuntimeError(
325
+ f"{self.base_url} did not answer within {self.timeout:.0f}s: {exc}"
326
+ ) from exc
327
+ choice = (data.get("choices") or [{}])[0]
328
+ usage = data.get("usage") or {}
329
+ finish = choice.get("finish_reason")
330
+ return Completion(
331
+ text=str((choice.get("message") or {}).get("content") or ""),
332
+ input_tokens=int(usage.get("prompt_tokens") or 0),
333
+ # Reasoning tokens are billed as output and are inside this count.
334
+ output_tokens=int(usage.get("completion_tokens") or 0),
335
+ model=str(data.get("model") or model),
336
+ stop_reason={"stop": "end_turn", "length": "max_tokens"}.get(
337
+ finish, finish
338
+ ),
339
+ raw=data,
340
+ )
341
+
342
+
168
343
  CannedResponse = Union[str, Completion]
169
344
 
170
345
 
@@ -252,8 +427,20 @@ class FakeClient:
252
427
  )
253
428
 
254
429
 
255
- def make_client(fake_responses: Optional[Any] = None) -> Client:
256
- """Build the client the CLI uses: a fake when canned responses are given."""
430
+ def make_client(
431
+ fake_responses: Optional[Any] = None, model: Optional[str] = None
432
+ ) -> Client:
433
+ """Build the client the CLI uses: a fake when canned responses are given,
434
+ otherwise the provider the model id belongs to (Anthropic by default)."""
257
435
  if fake_responses is not None:
258
436
  return FakeClient(fake_responses)
259
- return AnthropicClient()
437
+ provider = provider_for(model or "")
438
+ if provider is None:
439
+ return AnthropicClient()
440
+ base_url, key_env = provider
441
+ api_key = os.environ.get(key_env)
442
+ if not api_key and "localhost" not in base_url and "127.0.0.1" not in base_url:
443
+ raise RuntimeError(
444
+ f"Model {model!r} is served by {base_url}; set {key_env} to use it."
445
+ )
446
+ return OpenAICompatClient(base_url, api_key)