rulesmith 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (173) hide show
  1. rulesmith-0.1.0/.gitignore +11 -0
  2. rulesmith-0.1.0/LICENSE +21 -0
  3. rulesmith-0.1.0/PKG-INFO +131 -0
  4. rulesmith-0.1.0/README.md +87 -0
  5. rulesmith-0.1.0/benchmarks/charts.py +312 -0
  6. rulesmith-0.1.0/benchmarks/chess.sh +86 -0
  7. rulesmith-0.1.0/benchmarks/doom.sh +104 -0
  8. rulesmith-0.1.0/benchmarks/escalation.py +117 -0
  9. rulesmith-0.1.0/benchmarks/escalation.sh +41 -0
  10. rulesmith-0.1.0/benchmarks/fitness.py +250 -0
  11. rulesmith-0.1.0/benchmarks/map.py +122 -0
  12. rulesmith-0.1.0/benchmarks/overnight.sh +166 -0
  13. rulesmith-0.1.0/benchmarks/results/banking77.json +6078 -0
  14. rulesmith-0.1.0/benchmarks/results/chess-stockfish-skill0.json +92 -0
  15. rulesmith-0.1.0/benchmarks/results/clothing.json +8058 -0
  16. rulesmith-0.1.0/benchmarks/results/escalation.json +5695 -0
  17. rulesmith-0.1.0/benchmarks/results/games.json +489 -0
  18. rulesmith-0.1.0/benchmarks/results/instruct/banking77.json +3263 -0
  19. rulesmith-0.1.0/benchmarks/results/instruct/clothing.json +3241 -0
  20. rulesmith-0.1.0/benchmarks/results/instruct/jobs.json +3393 -0
  21. rulesmith-0.1.0/benchmarks/results/instruct/news.json +3733 -0
  22. rulesmith-0.1.0/benchmarks/results/instruct/trec.json +3050 -0
  23. rulesmith-0.1.0/benchmarks/results/jobs.json +7935 -0
  24. rulesmith-0.1.0/benchmarks/results/news.json +7457 -0
  25. rulesmith-0.1.0/benchmarks/results/training-terms.json +342 -0
  26. rulesmith-0.1.0/benchmarks/results/trec.json +6807 -0
  27. rulesmith-0.1.0/benchmarks/results/tuned/clothing.json +3243 -0
  28. rulesmith-0.1.0/benchmarks/results/tuned/jobs.json +3409 -0
  29. rulesmith-0.1.0/benchmarks/results/tuned/news.json +3700 -0
  30. rulesmith-0.1.0/benchmarks/results/tuned/trec.json +3050 -0
  31. rulesmith-0.1.0/benchmarks/tuned-judge.sh +53 -0
  32. rulesmith-0.1.0/benchmarks/tuned_judge.py +106 -0
  33. rulesmith-0.1.0/docs/benchmarks.md +496 -0
  34. rulesmith-0.1.0/docs/chess.md +258 -0
  35. rulesmith-0.1.0/docs/cli.md +219 -0
  36. rulesmith-0.1.0/docs/doom.md +474 -0
  37. rulesmith-0.1.0/docs/graphs.md +369 -0
  38. rulesmith-0.1.0/docs/maps.md +196 -0
  39. rulesmith-0.1.0/docs/optimization.md +192 -0
  40. rulesmith-0.1.0/docs/python.md +98 -0
  41. rulesmith-0.1.0/docs/quickstart.md +334 -0
  42. rulesmith-0.1.0/docs/related-work.md +94 -0
  43. rulesmith-0.1.0/docs/rules.md +695 -0
  44. rulesmith-0.1.0/docs/tasks.md +220 -0
  45. rulesmith-0.1.0/examples/README.md +25 -0
  46. rulesmith-0.1.0/examples/chess/duel.json +51 -0
  47. rulesmith-0.1.0/examples/chess/graphs/always-capture.rules +3 -0
  48. rulesmith-0.1.0/examples/chess/graphs/always-check.rules +1 -0
  49. rulesmith-0.1.0/examples/chess/graphs/always-quiet.rules +1 -0
  50. rulesmith-0.1.0/examples/chess/graphs/developer.rules +5 -0
  51. rulesmith-0.1.0/examples/chess/graphs/greedy.rules +9 -0
  52. rulesmith-0.1.0/examples/chess/graphs/hunter.rules +14 -0
  53. rulesmith-0.1.0/examples/chess/graphs/league-best.json +144 -0
  54. rulesmith-0.1.0/examples/chess/graphs/searched.json +212 -0
  55. rulesmith-0.1.0/examples/chess/league.json +147 -0
  56. rulesmith-0.1.0/examples/chess/stockfish.json +80 -0
  57. rulesmith-0.1.0/examples/doom/basic.json +10 -0
  58. rulesmith-0.1.0/examples/doom/defend_the_center.json +34 -0
  59. rulesmith-0.1.0/examples/doom/defend_the_line.json +35 -0
  60. rulesmith-0.1.0/examples/doom/duel-maps.json +196 -0
  61. rulesmith-0.1.0/examples/doom/duel.json +192 -0
  62. rulesmith-0.1.0/examples/doom/e1m1.json +82 -0
  63. rulesmith-0.1.0/examples/doom/freedoom.json +220 -0
  64. rulesmith-0.1.0/examples/doom/graphs/duel-attack.json +1 -0
  65. rulesmith-0.1.0/examples/doom/graphs/duel-hunter.json +782 -0
  66. rulesmith-0.1.0/examples/doom/graphs/duel-jev.json +702 -0
  67. rulesmith-0.1.0/examples/doom/graphs/duel-searched.json +471 -0
  68. rulesmith-0.1.0/examples/doom/graphs/e1m1-exit.json +517 -0
  69. rulesmith-0.1.0/examples/doom/graphs/e1m1-jev-seed.json +577 -0
  70. rulesmith-0.1.0/examples/doom/graphs/e1m1-reads.json +267 -0
  71. rulesmith-0.1.0/examples/doom/graphs/e1m1-searched.json +595 -0
  72. rulesmith-0.1.0/examples/doom/graphs/e1m1-seed.json +201 -0
  73. rulesmith-0.1.0/examples/doom/graphs/e1m1-target.json +339 -0
  74. rulesmith-0.1.0/examples/doom/health_gathering.json +34 -0
  75. rulesmith-0.1.0/examples/doom/health_gathering_supreme.json +34 -0
  76. rulesmith-0.1.0/examples/doom/predict_position.json +32 -0
  77. rulesmith-0.1.0/examples/doom/rocket_basic.json +32 -0
  78. rulesmith-0.1.0/examples/doom/simpler_basic.json +32 -0
  79. rulesmith-0.1.0/examples/doom/take_cover.json +32 -0
  80. rulesmith-0.1.0/examples/order-review/graph.rules +10 -0
  81. rulesmith-0.1.0/examples/reply-check/graph.rules +7 -0
  82. rulesmith-0.1.0/examples/support-routing/dag.json +147 -0
  83. rulesmith-0.1.0/examples/support-routing/evaluation.json +158 -0
  84. rulesmith-0.1.0/examples/support-routing/graph.rules +5 -0
  85. rulesmith-0.1.0/examples/support-routing/input.json +3 -0
  86. rulesmith-0.1.0/examples/support-routing/seed.json +27 -0
  87. rulesmith-0.1.0/examples/support-routing/task.json +70 -0
  88. rulesmith-0.1.0/examples/ticket-handling/map.json +47 -0
  89. rulesmith-0.1.0/examples/ticket-handling/units/billing.rules +5 -0
  90. rulesmith-0.1.0/examples/ticket-triage/evaluation.json +422 -0
  91. rulesmith-0.1.0/examples/ticket-triage/graph.rules +6 -0
  92. rulesmith-0.1.0/examples/ticket-triage/task.json +415 -0
  93. rulesmith-0.1.0/examples/trust-safety/evaluation.json +7682 -0
  94. rulesmith-0.1.0/examples/trust-safety/map.json +58 -0
  95. rulesmith-0.1.0/examples/trust-safety/task.json +1868 -0
  96. rulesmith-0.1.0/examples/trust-safety/units/intents.rules +1 -0
  97. rulesmith-0.1.0/examples/trust-safety/units/postings.rules +7 -0
  98. rulesmith-0.1.0/examples/trust-safety/units/reviews.rules +4 -0
  99. rulesmith-0.1.0/pyproject.toml +97 -0
  100. rulesmith-0.1.0/src/rulesmith/__init__.py +1 -0
  101. rulesmith-0.1.0/src/rulesmith/ablate.py +99 -0
  102. rulesmith-0.1.0/src/rulesmith/arena.py +172 -0
  103. rulesmith-0.1.0/src/rulesmith/bench.py +1068 -0
  104. rulesmith-0.1.0/src/rulesmith/calibrate.py +457 -0
  105. rulesmith-0.1.0/src/rulesmith/chat_judge.py +205 -0
  106. rulesmith-0.1.0/src/rulesmith/chess.py +526 -0
  107. rulesmith-0.1.0/src/rulesmith/clef.py +66 -0
  108. rulesmith-0.1.0/src/rulesmith/cli.py +1234 -0
  109. rulesmith-0.1.0/src/rulesmith/diagram.py +226 -0
  110. rulesmith-0.1.0/src/rulesmith/doom.py +550 -0
  111. rulesmith-0.1.0/src/rulesmith/extract.py +77 -0
  112. rulesmith-0.1.0/src/rulesmith/grade.py +85 -0
  113. rulesmith-0.1.0/src/rulesmith/graph.py +975 -0
  114. rulesmith-0.1.0/src/rulesmith/label.py +67 -0
  115. rulesmith-0.1.0/src/rulesmith/level.py +389 -0
  116. rulesmith-0.1.0/src/rulesmith/maps.py +96 -0
  117. rulesmith-0.1.0/src/rulesmith/mine.py +313 -0
  118. rulesmith-0.1.0/src/rulesmith/optimize.py +931 -0
  119. rulesmith-0.1.0/src/rulesmith/rules.py +1017 -0
  120. rulesmith-0.1.0/src/rulesmith/runtime.py +711 -0
  121. rulesmith-0.1.0/src/rulesmith/serve.py +68 -0
  122. rulesmith-0.1.0/src/rulesmith/tuning.py +134 -0
  123. rulesmith-0.1.0/tests/conftest.py +212 -0
  124. rulesmith-0.1.0/tests/test_ablate.py +120 -0
  125. rulesmith-0.1.0/tests/test_arena.py +394 -0
  126. rulesmith-0.1.0/tests/test_arithmetic.py +59 -0
  127. rulesmith-0.1.0/tests/test_bench.py +342 -0
  128. rulesmith-0.1.0/tests/test_briefing.py +98 -0
  129. rulesmith-0.1.0/tests/test_calibrate.py +390 -0
  130. rulesmith-0.1.0/tests/test_charts.py +39 -0
  131. rulesmith-0.1.0/tests/test_chat_judge.py +235 -0
  132. rulesmith-0.1.0/tests/test_chess.py +256 -0
  133. rulesmith-0.1.0/tests/test_clef.py +207 -0
  134. rulesmith-0.1.0/tests/test_cli.py +749 -0
  135. rulesmith-0.1.0/tests/test_compose.py +163 -0
  136. rulesmith-0.1.0/tests/test_conditions.py +242 -0
  137. rulesmith-0.1.0/tests/test_confidence.py +57 -0
  138. rulesmith-0.1.0/tests/test_diagram.py +122 -0
  139. rulesmith-0.1.0/tests/test_doom.py +454 -0
  140. rulesmith-0.1.0/tests/test_doom_level.py +315 -0
  141. rulesmith-0.1.0/tests/test_drift.py +86 -0
  142. rulesmith-0.1.0/tests/test_examples.py +51 -0
  143. rulesmith-0.1.0/tests/test_extract.py +184 -0
  144. rulesmith-0.1.0/tests/test_fitness.py +101 -0
  145. rulesmith-0.1.0/tests/test_given.py +84 -0
  146. rulesmith-0.1.0/tests/test_grade.py +185 -0
  147. rulesmith-0.1.0/tests/test_graph.py +376 -0
  148. rulesmith-0.1.0/tests/test_integration.py +1041 -0
  149. rulesmith-0.1.0/tests/test_judges.py +320 -0
  150. rulesmith-0.1.0/tests/test_label.py +70 -0
  151. rulesmith-0.1.0/tests/test_label_lists.py +151 -0
  152. rulesmith-0.1.0/tests/test_lazy.py +59 -0
  153. rulesmith-0.1.0/tests/test_length.py +50 -0
  154. rulesmith-0.1.0/tests/test_lists.py +110 -0
  155. rulesmith-0.1.0/tests/test_live.py +59 -0
  156. rulesmith-0.1.0/tests/test_map.py +300 -0
  157. rulesmith-0.1.0/tests/test_map_demo.py +42 -0
  158. rulesmith-0.1.0/tests/test_measured.py +180 -0
  159. rulesmith-0.1.0/tests/test_mine.py +226 -0
  160. rulesmith-0.1.0/tests/test_openai_judge.py +64 -0
  161. rulesmith-0.1.0/tests/test_outputs.py +62 -0
  162. rulesmith-0.1.0/tests/test_proposer.py +60 -0
  163. rulesmith-0.1.0/tests/test_resume.py +142 -0
  164. rulesmith-0.1.0/tests/test_round_trip.py +146 -0
  165. rulesmith-0.1.0/tests/test_rules.py +381 -0
  166. rulesmith-0.1.0/tests/test_ruling.py +188 -0
  167. rulesmith-0.1.0/tests/test_runtime.py +317 -0
  168. rulesmith-0.1.0/tests/test_scoring.py +185 -0
  169. rulesmith-0.1.0/tests/test_serve.py +91 -0
  170. rulesmith-0.1.0/tests/test_tuning.py +194 -0
  171. rulesmith-0.1.0/tests/test_unit_search.py +222 -0
  172. rulesmith-0.1.0/tests/test_vote.py +119 -0
  173. rulesmith-0.1.0/uv.lock +2867 -0
@@ -0,0 +1,11 @@
1
+ .venv/
2
+ __pycache__/
3
+ .pytest_cache/
4
+ .hypothesis/
5
+ .ruff_cache/
6
+ *.egg-info/
7
+ dist/
8
+ runs/
9
+ .env
10
+ .env.*
11
+ .DS_Store
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 rulesmith contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in
13
+ all copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
21
+ THE SOFTWARE.
@@ -0,0 +1,131 @@
1
+ Metadata-Version: 2.5
2
+ Name: rulesmith
3
+ Version: 0.1.0
4
+ Summary: Search for rule programs that ask a small judge for typed decisions, with DSPy and GEPA.
5
+ Project-URL: Homepage, https://github.com/bradAGI/rulesmith
6
+ Project-URL: Repository, https://github.com/bradAGI/rulesmith
7
+ Project-URL: Issues, https://github.com/bradAGI/rulesmith/issues
8
+ Author: rulesmith contributors
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: decision graphs,dspy,gepa,llm,program search,rules
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Programming Language :: Python :: 3.14
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Requires-Python: <3.15,>=3.12
21
+ Requires-Dist: dspy==3.4.0
22
+ Requires-Dist: gepa==0.1.4
23
+ Requires-Dist: httpx<1,>=0.28
24
+ Requires-Dist: pydantic<3,>=2.11
25
+ Requires-Dist: typesafe-sdk<0.7,>=0.6
26
+ Provides-Extra: bench
27
+ Requires-Dist: huggingface-hub<2,>=1; extra == 'bench'
28
+ Requires-Dist: pyarrow<24,>=17; extra == 'bench'
29
+ Requires-Dist: scikit-learn<2,>=1.5; extra == 'bench'
30
+ Provides-Extra: chess
31
+ Requires-Dist: chess<2,>=1.11; extra == 'chess'
32
+ Provides-Extra: clef
33
+ Requires-Dist: huggingface-hub<2,>=1; extra == 'clef'
34
+ Requires-Dist: pillow<13,>=11; extra == 'clef'
35
+ Requires-Dist: torch<3,>=2.11; extra == 'clef'
36
+ Requires-Dist: torchvision<1,>=0.26; extra == 'clef'
37
+ Requires-Dist: transformers[torch]<6,>=5.10.2; extra == 'clef'
38
+ Provides-Extra: doom
39
+ Requires-Dist: imageio-ffmpeg<1,>=0.6; extra == 'doom'
40
+ Requires-Dist: numpy<3,>=2; extra == 'doom'
41
+ Requires-Dist: pillow<13,>=11; extra == 'doom'
42
+ Requires-Dist: vizdoom<2,>=1.3; extra == 'doom'
43
+ Description-Content-Type: text/markdown
44
+
45
+ # rulesmith
46
+
47
+ ![rulesmith: rules that ask a local model for typed decisions](https://raw.githubusercontent.com/bradAGI/rulesmith/main/image.png)
48
+
49
+ **rulesmith lets DSPy build decision graphs for you.** Give it a classification
50
+ task and labeled examples. It searches for a small program of rules: cheap tests
51
+ on the input's fields settle the cases they can, and a local or hosted judge is
52
+ asked typed questions about the rest. [GEPA](https://github.com/gepa-ai/gepa)
53
+ keeps the revisions that score better, and the result is a rules file you can
54
+ read, edit, run, and serve.
55
+
56
+ ```
57
+ read has_company_logo in [0, 1]
58
+ read length of company_profile in [0, 10000]
59
+ when has_company_logo == 0 and length of company_profile == 0: fraudulent
60
+ always: ask "Is this job posting a scam?" [fraudulent: "fraudulent"] [genuine: "genuine"]
61
+ ```
62
+
63
+ ![Same 4B judge, same 100 examples: rulesmith beat DSPy's tuned classifier by 6.7 points on job-fraud detection with 29% of the model calls, and tied it on clothing reviews with 6%](https://raw.githubusercontent.com/bradAGI/rulesmith/main/docs/assets/claim.svg)
64
+
65
+ The [benchmarks](https://github.com/bradAGI/rulesmith/blob/main/docs/benchmarks.md) have every dataset, including where rulesmith
66
+ does not help: on one-sentence inputs it ties DSPy, and on job postings a TF-IDF
67
+ classifier still beats it.
68
+
69
+ An independent research prototype, not affiliated with TypeSafe.
70
+
71
+ ## Installation
72
+
73
+ ```sh
74
+ pip install rulesmith
75
+ ```
76
+
77
+ Python 3.12 to 3.14. Extras for the games and benchmarks are listed in the
78
+ [quickstart](https://github.com/bradAGI/rulesmith/blob/main/docs/quickstart.md#installation).
79
+
80
+ ## Quickstart
81
+
82
+ With a judge running ([Ruling](https://github.com/bradAGI/ruling) on a Mac,
83
+ Cloudflare's Clef-flash on an NVIDIA GPU or a Mac, TypeSafe's hosted Jev, or any
84
+ OpenAI-compatible chat API that returns log-probabilities), any OpenAI-compatible model to write the rules, and a
85
+ labeled task such as [this one](https://github.com/bradAGI/rulesmith/blob/main/examples/support-routing/task.json):
86
+
87
+ ```sh
88
+ export OPENAI_API_KEY=local-no-auth # a local model server takes any key
89
+ rulesmith optimize task.json --backend ruling --max-workers 1 \
90
+ --reflection-model "openai/<model>" --reflection-base-url http://127.0.0.1:8080/v1 \
91
+ --output runs/my-task
92
+ rulesmith run runs/my-task/plan.rules input.json --backend ruling
93
+ ```
94
+
95
+ The [quickstart](https://github.com/bradAGI/rulesmith/blob/main/docs/quickstart.md) walks through it with every command's output.
96
+
97
+ ## Beyond classification
98
+
99
+ The same search plays games. Pitted against a league of other graphs, this one learned to
100
+ hunt in Doom deathmatch and beat the champion 12-0, with no model calls at all
101
+ ([how](https://github.com/bradAGI/rulesmith/blob/main/docs/doom.md)):
102
+
103
+ ![A searched rules graph hunting down the Doom champion it dethroned](https://raw.githubusercontent.com/bradAGI/rulesmith/main/docs/assets/doom/duel-hunter.gif)
104
+
105
+ ## Learn more
106
+
107
+ - [Quickstart](https://github.com/bradAGI/rulesmith/blob/main/docs/quickstart.md): installation, judges, and a task to a running graph.
108
+ - [Notebook](https://github.com/bradAGI/rulesmith/blob/main/notebooks/quickstart.ipynb): rules for spotting fake job postings, end to end in Colab with an OpenAI key.
109
+ - [Local notebook](https://github.com/bradAGI/rulesmith/blob/main/notebooks/local.ipynb): the same with Clef-flash judging on your Mac or NVIDIA GPU, free per call.
110
+ - [Tasks](https://github.com/bradAGI/rulesmith/blob/main/docs/tasks.md): the task file, drafting labels, structured inputs, costs, and graders.
111
+ - [Rules](https://github.com/bradAGI/rulesmith/blob/main/docs/rules.md): the language search writes, confidence, calibration, escalation, and extraction.
112
+ - [Graph format](https://github.com/bradAGI/rulesmith/blob/main/docs/graphs.md): the JSON behind each node, execution, and drawing a graph.
113
+ - [Maps](https://github.com/bradAGI/rulesmith/blob/main/docs/maps.md): many decisions as one map, searched a unit at a time.
114
+ - [Optimization](https://github.com/bradAGI/rulesmith/blob/main/docs/optimization.md): how search runs, its settings, and its limits.
115
+ - [Command line](https://github.com/bradAGI/rulesmith/blob/main/docs/cli.md): every command, including `serve` and `drift`, and the backends.
116
+ - [Python](https://github.com/bradAGI/rulesmith/blob/main/docs/python.md): running and building graphs from code.
117
+ - [Benchmarks](https://github.com/bradAGI/rulesmith/blob/main/docs/benchmarks.md): rulesmith against DSPy, TF-IDF and direct calls on five public datasets.
118
+ - [Related work](https://github.com/bradAGI/rulesmith/blob/main/docs/related-work.md): when to use rulesmith and when DSPy, and what came before.
119
+ - [Doom](https://github.com/bradAGI/rulesmith/blob/main/docs/doom.md) and [Chess](https://github.com/bradAGI/rulesmith/blob/main/docs/chess.md): graphs that play.
120
+ - [Claude Code skill](https://github.com/bradAGI/rulesmith/blob/main/skills/rulesmith/SKILL.md): copy `skills/rulesmith` into `~/.claude/skills/`.
121
+
122
+ ## Development
123
+
124
+ ```sh
125
+ uv sync --locked
126
+ uv run pytest -q
127
+ ```
128
+
129
+ ## License
130
+
131
+ [MIT](https://github.com/bradAGI/rulesmith/blob/main/LICENSE).
@@ -0,0 +1,87 @@
1
+ # rulesmith
2
+
3
+ ![rulesmith: rules that ask a local model for typed decisions](image.png)
4
+
5
+ **rulesmith lets DSPy build decision graphs for you.** Give it a classification
6
+ task and labeled examples. It searches for a small program of rules: cheap tests
7
+ on the input's fields settle the cases they can, and a local or hosted judge is
8
+ asked typed questions about the rest. [GEPA](https://github.com/gepa-ai/gepa)
9
+ keeps the revisions that score better, and the result is a rules file you can
10
+ read, edit, run, and serve.
11
+
12
+ ```
13
+ read has_company_logo in [0, 1]
14
+ read length of company_profile in [0, 10000]
15
+ when has_company_logo == 0 and length of company_profile == 0: fraudulent
16
+ always: ask "Is this job posting a scam?" [fraudulent: "fraudulent"] [genuine: "genuine"]
17
+ ```
18
+
19
+ ![Same 4B judge, same 100 examples: rulesmith beat DSPy's tuned classifier by 6.7 points on job-fraud detection with 29% of the model calls, and tied it on clothing reviews with 6%](docs/assets/claim.svg)
20
+
21
+ The [benchmarks](docs/benchmarks.md) have every dataset, including where rulesmith
22
+ does not help: on one-sentence inputs it ties DSPy, and on job postings a TF-IDF
23
+ classifier still beats it.
24
+
25
+ An independent research prototype, not affiliated with TypeSafe.
26
+
27
+ ## Installation
28
+
29
+ ```sh
30
+ pip install rulesmith
31
+ ```
32
+
33
+ Python 3.12 to 3.14. Extras for the games and benchmarks are listed in the
34
+ [quickstart](docs/quickstart.md#installation).
35
+
36
+ ## Quickstart
37
+
38
+ With a judge running ([Ruling](https://github.com/bradAGI/ruling) on a Mac,
39
+ Cloudflare's Clef-flash on an NVIDIA GPU or a Mac, TypeSafe's hosted Jev, or any
40
+ OpenAI-compatible chat API that returns log-probabilities), any OpenAI-compatible model to write the rules, and a
41
+ labeled task such as [this one](examples/support-routing/task.json):
42
+
43
+ ```sh
44
+ export OPENAI_API_KEY=local-no-auth # a local model server takes any key
45
+ rulesmith optimize task.json --backend ruling --max-workers 1 \
46
+ --reflection-model "openai/<model>" --reflection-base-url http://127.0.0.1:8080/v1 \
47
+ --output runs/my-task
48
+ rulesmith run runs/my-task/plan.rules input.json --backend ruling
49
+ ```
50
+
51
+ The [quickstart](docs/quickstart.md) walks through it with every command's output.
52
+
53
+ ## Beyond classification
54
+
55
+ The same search plays games. Pitted against a league of other graphs, this one learned to
56
+ hunt in Doom deathmatch and beat the champion 12-0, with no model calls at all
57
+ ([how](docs/doom.md)):
58
+
59
+ ![A searched rules graph hunting down the Doom champion it dethroned](docs/assets/doom/duel-hunter.gif)
60
+
61
+ ## Learn more
62
+
63
+ - [Quickstart](docs/quickstart.md): installation, judges, and a task to a running graph.
64
+ - [Notebook](notebooks/quickstart.ipynb): rules for spotting fake job postings, end to end in Colab with an OpenAI key.
65
+ - [Local notebook](notebooks/local.ipynb): the same with Clef-flash judging on your Mac or NVIDIA GPU, free per call.
66
+ - [Tasks](docs/tasks.md): the task file, drafting labels, structured inputs, costs, and graders.
67
+ - [Rules](docs/rules.md): the language search writes, confidence, calibration, escalation, and extraction.
68
+ - [Graph format](docs/graphs.md): the JSON behind each node, execution, and drawing a graph.
69
+ - [Maps](docs/maps.md): many decisions as one map, searched a unit at a time.
70
+ - [Optimization](docs/optimization.md): how search runs, its settings, and its limits.
71
+ - [Command line](docs/cli.md): every command, including `serve` and `drift`, and the backends.
72
+ - [Python](docs/python.md): running and building graphs from code.
73
+ - [Benchmarks](docs/benchmarks.md): rulesmith against DSPy, TF-IDF and direct calls on five public datasets.
74
+ - [Related work](docs/related-work.md): when to use rulesmith and when DSPy, and what came before.
75
+ - [Doom](docs/doom.md) and [Chess](docs/chess.md): graphs that play.
76
+ - [Claude Code skill](skills/rulesmith/SKILL.md): copy `skills/rulesmith` into `~/.claude/skills/`.
77
+
78
+ ## Development
79
+
80
+ ```sh
81
+ uv sync --locked
82
+ uv run pytest -q
83
+ ```
84
+
85
+ ## License
86
+
87
+ [MIT](LICENSE).
@@ -0,0 +1,312 @@
1
+ """The benchmark charts in the docs, drawn from the results files.
2
+
3
+ python3 benchmarks/charts.py
4
+
5
+ Writes docs/assets/benchmarks.svg, accuracy against judge calls per row for rulesmith's search
6
+ and the baselines on each dataset; docs/assets/escalation.svg, accuracy against the share of
7
+ rows sent to the larger judge; and docs/assets/claim.svg, the README's headline: search against
8
+ DSPy's typed classifier on the same tuned judge. Every number is a mean over the seeds in
9
+ benchmarks/results; the tables in docs/benchmarks.md carry the spread and the tests of
10
+ significance.
11
+ """
12
+
13
+ import json
14
+ from pathlib import Path
15
+ from statistics import mean
16
+
17
+ ROOT = Path(__file__).resolve().parents[1]
18
+ RESULTS = ROOT / "benchmarks" / "results"
19
+ ASSETS = ROOT / "docs" / "assets"
20
+
21
+ # Where rules over fields help first, then where the input is one piece of free text.
22
+ DATASETS = ("clothing", "jobs", "news", "trec", "banking77")
23
+ # The baselines that ask a language model on every row; the best of them per dataset is the
24
+ # fair comparison, so search is never shown beside a weak one.
25
+ ASKING = {
26
+ "direct": "direct call",
27
+ "dspy": "DSPy",
28
+ "dspy-cot": "DSPy, reasoning",
29
+ "dspy-typed": "DSPy typed",
30
+ }
31
+ # The README's headline: where the input has fields, on the judge tuned on each seed's examples.
32
+ CLAIM = (
33
+ ("jobs", "Job postings: scam or genuine?"),
34
+ ("clothing", "Clothing reviews: recommend or not?"),
35
+ )
36
+ ESCALATION = (("small", "4B alone"), ("large", "Clef-flash alone"), ("escalate", "escalating"))
37
+
38
+ # rulesmith is the one hue; every baseline is context, in the neutral, and named beside its mark.
39
+ STYLE = """
40
+ .bg { fill: #fcfcfb } .ink { fill: #0b0b0b } .muted { fill: #52514e }
41
+ .grid { stroke: #e4e3df } .axis { stroke: #b5b4ae }
42
+ .us { fill: #2a78d6 } .them { fill: #6b6a66 } .ring { stroke: #fcfcfb }
43
+ .mix { stroke: #6b6a66 }
44
+ .halo { stroke: #fcfcfb; stroke-width: 4px; paint-order: stroke; stroke-linejoin: round }
45
+ @media (prefers-color-scheme: dark) {
46
+ .bg { fill: #1a1a19 } .ink { fill: #ffffff } .muted { fill: #c3c2b7 }
47
+ .grid { stroke: #33322f } .axis { stroke: #57564f }
48
+ .us { fill: #3987e5 } .them { fill: #8f8e89 } .ring { stroke: #1a1a19 }
49
+ .mix { stroke: #8f8e89 }
50
+ .halo { stroke: #1a1a19 }
51
+ }
52
+ """
53
+
54
+ PANEL, WIDE, GAP, LEFT, TOP, PLOT = 176, 300, 62, 52, 74, 190
55
+ FONT = 'font-family="system-ui, sans-serif" font-size="12"'
56
+
57
+
58
+ def by_method(dataset: str) -> dict[str, tuple[float, float]]:
59
+ """Mean accuracy and judge calls per row of each method on a dataset, over its seeds."""
60
+ runs = json.loads((RESULTS / f"{dataset}.json").read_text())["runs"]
61
+ methods = {r["method"] for r in runs}
62
+ return {
63
+ m: (
64
+ mean(r["accuracy"] for r in runs if r["method"] == m),
65
+ mean(r["calls_per_example"] for r in runs if r["method"] == m),
66
+ )
67
+ for m in methods
68
+ }
69
+
70
+
71
+ def comparison(found: dict[str, tuple[float, float]]) -> list[tuple[str, float, float, bool]]:
72
+ """What a panel shows: search, the best baseline that asks the model on every row, and the
73
+ trained classifier that never asks, as (name, calls, accuracy, is rulesmith)."""
74
+ best = max((m for m in ASKING if m in found), key=lambda m: found[m][0])
75
+ return [
76
+ ("TF-IDF", found["tfidf"][1], found["tfidf"][0], False),
77
+ (ASKING[best], found[best][1], found[best][0], False),
78
+ ("rulesmith", found["search"][1], found["search"][0], True),
79
+ ]
80
+
81
+
82
+ def escalated() -> dict[str, list[tuple[str, float, float]]]:
83
+ """Per dataset, each way of judging as (name, larger-judge calls per row, accuracy)."""
84
+ runs = json.loads((RESULTS / "escalation.json").read_text())["runs"]
85
+ found = {}
86
+ for dataset in sorted({r["dataset"] for r in runs}):
87
+ found[dataset] = []
88
+ for graph, name in ESCALATION:
89
+ mine = [r for r in runs if r["dataset"] == dataset and r["graph"] == graph]
90
+ large = mean(r["calls_per_row"]["large"] for r in mine)
91
+ found[dataset].append((name, large, mean(r["accuracy"] for r in mine)))
92
+ return found
93
+
94
+
95
+ def tuned(dataset: str, method: str) -> list[dict]:
96
+ """A method's runs on the tuned judge, seed by seed, so pooled rows line up across methods."""
97
+ runs = json.loads((RESULTS / "tuned" / f"{dataset}.json").read_text())["runs"]
98
+ return sorted((r for r in runs if r["method"] == method), key=lambda r: r["seed"])
99
+
100
+
101
+ def p_value(p: float) -> str:
102
+ """A p-value as a reader writes it: 0.66, or 3×10⁻⁶ rather than 3.2e-06."""
103
+ if p >= 0.001:
104
+ return f"{p:.2g}"
105
+ mantissa, exponent = f"{p:.0e}".split("e")
106
+ raised = str(int(exponent)).translate(str.maketrans("-0123456789", "⁻⁰¹²³⁴⁵⁶⁷⁸⁹"))
107
+ return f"{mantissa}×10{raised}"
108
+
109
+
110
+ def ticks(low: float, high: float) -> tuple[float, float, list[float]]:
111
+ """A y range padded to the nearest 0.05 around the values, and its gridlines."""
112
+ step = 0.05 if high - low < 0.25 else 0.1
113
+ bottom = max(0.0, (int(low / step) - (0 if low % step else 1)) * step)
114
+ top = min(1.0, (int(high / step) + 1) * step)
115
+ count = round((top - bottom) / step)
116
+ return bottom, top, [round(bottom + i * step, 2) for i in range(count + 1)]
117
+
118
+
119
+ def labels(points: list[tuple[float, float]]) -> list[int]:
120
+ """Which way each point's label goes: up, unless a higher point sits near enough that the
121
+ labels would meet, in which case the lower one is labeled below."""
122
+ sides = [-1] * len(points)
123
+ for i, (x, y) in enumerate(points):
124
+ for j, (x2, y2) in enumerate(points):
125
+ if i != j and abs(x - x2) < 70 and 0 <= y - y2 < 26:
126
+ sides[i] = 1
127
+ return sides
128
+
129
+
130
+ def panel(left: float, title: str, points, x_label: str, segment=None, size=PANEL) -> list[str]:
131
+ """One small multiple: its own y scale, x from 0 to 1, each point named beside it."""
132
+ values = [y for _, _, y, *_ in points]
133
+ bottom, top, grid = ticks(min(values), max(values))
134
+ x = lambda v: left + 10 + v * (size - 20) # noqa: E731
135
+ y = lambda v: TOP + PLOT - (v - bottom) / (top - bottom) * PLOT # noqa: E731
136
+ out = [f'<text x="{left}" y="{TOP - 14}" font-weight="600" class="ink">{title}</text>']
137
+ for g in grid:
138
+ out.append(
139
+ f'<line x1="{left}" y1="{y(g):.1f}" x2="{left + size}" y2="{y(g):.1f}" class="grid"/>'
140
+ )
141
+ out.append(
142
+ f'<text x="{left - 6}" y="{y(g) + 4:.1f}" text-anchor="end" class="muted">'
143
+ f"{g:.2f}</text>"
144
+ )
145
+ base = TOP + PLOT
146
+ out.append(f'<line x1="{left}" y1="{base}" x2="{left + size}" y2="{base}" class="axis"/>')
147
+ for v in (0, 0.5, 1):
148
+ out.append(
149
+ f'<text x="{x(v):.1f}" y="{base + 16}" text-anchor="middle" class="muted">{v:g}</text>'
150
+ )
151
+ out.append(
152
+ f'<text x="{left + size / 2}" y="{base + 34}" text-anchor="middle" class="muted">'
153
+ f"{x_label}</text>"
154
+ )
155
+ if segment:
156
+ (x1, y1), (x2, y2) = segment
157
+ out.append(
158
+ f'<line x1="{x(x1):.1f}" y1="{y(y1):.1f}" x2="{x(x2):.1f}" y2="{y(y2):.1f}" '
159
+ 'class="mix" stroke-width="1.5" stroke-dasharray="4 4"/>'
160
+ )
161
+ placed = [(x(c), y(a)) for _, c, a, *_ in points]
162
+ for (name, calls, accuracy, *ours), (px, py), side in zip(
163
+ points, placed, labels(placed), strict=True
164
+ ):
165
+ kind = "us" if ours and ours[0] else "them"
166
+ anchor = "start" if px < left + 30 else "end" if px > left + size - 30 else "middle"
167
+ weight = ' font-weight="600"' if kind == "us" else ""
168
+ out.append(
169
+ f'<circle cx="{px:.1f}" cy="{py:.1f}" r="5" class="{kind} ring" stroke-width="2">'
170
+ f"<title>{name}: {accuracy:.3f} accuracy, {calls:.2f} calls per row</title></circle>"
171
+ )
172
+ ly = py - 10 if side < 0 else py + 19
173
+ out.append(
174
+ f'<text x="{px:.1f}" y="{ly:.1f}" text-anchor="{anchor}" class="ink halo"{weight}>'
175
+ f"{name} {accuracy:.3f}</text>"
176
+ )
177
+ return out
178
+
179
+
180
+ def svg(width: int, height: int, title: str, subtitle: str, body: list[str]) -> str:
181
+ return "\n".join(
182
+ [
183
+ f'<svg xmlns="http://www.w3.org/2000/svg" width="{width}" height="{height}" '
184
+ f'viewBox="0 0 {width} {height}" {FONT}>',
185
+ f"<style>{STYLE}</style>",
186
+ f'<rect width="{width}" height="{height}" class="bg"/>',
187
+ f'<text x="16" y="24" font-size="14" font-weight="600" class="ink">{title}</text>',
188
+ f'<text x="16" y="42" class="muted">{subtitle}</text>',
189
+ *body,
190
+ "</svg>",
191
+ "",
192
+ ]
193
+ )
194
+
195
+
196
+ def benchmarks_chart() -> str:
197
+ body = []
198
+ for i, dataset in enumerate(DATASETS):
199
+ shown = comparison(by_method(dataset))
200
+ body += panel(LEFT + i * (PANEL + GAP), dataset, shown, "judge calls per row")
201
+ width = LEFT + len(DATASETS) * (PANEL + GAP) - GAP + 24
202
+ return svg(
203
+ width,
204
+ TOP + PLOT + 52,
205
+ "Accuracy against judge calls per row, on 300 held-out rows",
206
+ "rulesmith's searched rules against the best baseline that asks on every row, and a "
207
+ "classifier that never asks. Means of three seeds; each panel has its own scale.",
208
+ body,
209
+ )
210
+
211
+
212
+ def escalation_chart() -> str:
213
+ body = []
214
+ for i, (dataset, points) in enumerate(escalated().items()):
215
+ marked = [(name, calls, acc, name == "escalating") for name, calls, acc in points]
216
+ ends = [(c, a) for n, c, a in points if n != "escalating"]
217
+ body += panel(
218
+ LEFT + i * (WIDE + GAP), dataset, marked, "larger-judge calls per row", ends, WIDE
219
+ )
220
+ width = LEFT + 2 * (WIDE + GAP) - GAP + 24
221
+ return svg(
222
+ width,
223
+ TOP + PLOT + 52,
224
+ "Escalating to a larger judge only where the small one is unsure",
225
+ "Above the dashed line is better than sending that share of rows to the larger judge at "
226
+ "random. Means of three seeds.",
227
+ body,
228
+ )
229
+
230
+
231
+ def bar(x: float, y: float, length: float, value: str, kind: str) -> list[str]:
232
+ """A bar from zero, and its value at the end of it in the text color."""
233
+ weight = ' font-weight="600"' if kind == "us" else ""
234
+ return [
235
+ f'<rect x="{x}" y="{y + 2}" width="{length:.1f}" height="18" rx="2" class="{kind}"/>',
236
+ f'<text x="{x + length + 6:.1f}" y="{y + 15}" class="ink"{weight}>{value}</text>',
237
+ ]
238
+
239
+
240
+ def claim_chart() -> str:
241
+ from rulesmith.bench import mcnemar
242
+
243
+ width, left, accuracy_width, calls_left, calls_width = 820, 200, 360, 640, 110
244
+ body = [
245
+ f'<text x="{left}" y="{TOP}" class="muted" font-weight="600">accuracy</text>',
246
+ f'<text x="{calls_left}" y="{TOP}" class="muted" font-weight="600">'
247
+ "model calls per row</text>",
248
+ ]
249
+ y = TOP + 34
250
+ for dataset, title in CLAIM:
251
+ ours, theirs = tuned(dataset, "search"), tuned(dataset, "dspy-typed")
252
+ pooled = [[c for r in runs for c in r["correct"]] for runs in (ours, theirs)]
253
+ p = mcnemar(*pooled)
254
+ rows = [
255
+ (
256
+ "rulesmith",
257
+ mean(r["accuracy"] for r in ours),
258
+ mean(r["calls_per_example"] for r in ours),
259
+ "us",
260
+ ),
261
+ (
262
+ "DSPy typed classifier",
263
+ mean(r["accuracy"] for r in theirs),
264
+ mean(r["calls_per_example"] for r in theirs),
265
+ "them",
266
+ ),
267
+ ]
268
+ gain = (rows[0][1] - rows[1][1]) * 100
269
+ verdict = f"+{gain:.1f} points" if p < 0.05 and gain > 0 else "tie"
270
+ share = rows[0][2] / rows[1][2]
271
+ body.append(
272
+ f'<text x="16" y="{y}" class="ink" font-size="14" font-weight="600">{title}</text>'
273
+ f'<text x="{left + 230}" y="{y}" class="muted">{verdict} · p = {p_value(p)} · '
274
+ f"{share:.0%} of the calls</text>"
275
+ )
276
+ y += 14
277
+ for name, accuracy, calls, kind in rows:
278
+ weight = ' font-weight="600"' if kind == "us" else ""
279
+ body += [
280
+ f'<text x="{left - 10}" y="{y + 15}" text-anchor="end" class="ink"{weight}>'
281
+ f"{name}</text>",
282
+ *bar(left, y, accuracy * accuracy_width, f"{accuracy:.3f}", kind),
283
+ *bar(calls_left, y, calls * calls_width, f"{calls:.2f}", kind),
284
+ ]
285
+ y += 26
286
+ y += 30
287
+ body.append(
288
+ f'<text x="16" y="{y}" class="muted">Means of 3 seeds, 300 held-out rows each; p from '
289
+ "exact McNemar tests over the pooled rows. Bars start at zero.</text>"
290
+ )
291
+ body.append(
292
+ f'<text x="16" y="{y + 18}" class="muted">On one-sentence inputs (TREC, Banking77) '
293
+ "rulesmith ties DSPy; on job postings a TF-IDF classifier scores higher still.</text>"
294
+ )
295
+ return svg(
296
+ width,
297
+ y + 36,
298
+ "Same 4B judge, same 100 examples, a fraction of the calls",
299
+ "rulesmith against DSPy's typed classifier (GEPA + ReAnchor), the judge fine-tuned on each "
300
+ "seed's examples.",
301
+ body,
302
+ )
303
+
304
+
305
+ def main() -> None:
306
+ (ASSETS / "benchmarks.svg").write_text(benchmarks_chart())
307
+ (ASSETS / "escalation.svg").write_text(escalation_chart())
308
+ (ASSETS / "claim.svg").write_text(claim_chart())
309
+
310
+
311
+ if __name__ == "__main__":
312
+ main()
@@ -0,0 +1,86 @@
1
+ #!/bin/zsh
2
+ # The chess searches against Stockfish behind docs/assets/fitness.svg: from the greedy graph,
3
+ # several search seeds at once, each with the same budget of games. A finished search is kept,
4
+ # so running this again picks up where an interrupted run stopped.
5
+ #
6
+ # benchmarks/chess.sh # three seeds, 1000 games each
7
+ # BUDGET=200 SEEDS="0" benchmarks/chess.sh
8
+ # kill $(cat runs/chess/driver.pid) # stop it, and only it
9
+ #
10
+ # Needs: Stockfish at $STOCKFISH, a proposer (a hosted model with its key in the environment,
11
+ # or a local server at $PROPOSER_URL), and Ruling at $JUDGE for any graph that asks it.
12
+ set -u
13
+ cd "${0:A:h}/.."
14
+
15
+ HOURS=${HOURS:-6}
16
+ BUDGET=${BUDGET:-1000}
17
+ SEEDS=(${=SEEDS:-0 1 2})
18
+ STOCKFISH=${STOCKFISH:-$HOME/.local/bin/stockfish}
19
+ JUDGE=${JUDGE:-http://127.0.0.1:8010}
20
+ PROPOSER_MODEL=${PROPOSER_MODEL:-openrouter/qwen/qwen3.8-27b}
21
+ PROPOSER_URL=${PROPOSER_URL-}
22
+ PROPOSER_OPTIONS=${PROPOSER_OPTIONS-'{"reasoning": {"max_tokens": 2048}}'}
23
+ TASK=examples/chess/stockfish.json
24
+ START=examples/chess/graphs/greedy.rules
25
+ LOGS=runs/chess
26
+ mkdir -p $LOGS benchmarks/results
27
+ export OPENAI_API_KEY=${OPENAI_API_KEY:-local-no-auth}
28
+
29
+ proposer=(--reflection-model $PROPOSER_MODEL --reflection-max-tokens 16384 --reflection-timeout 3600)
30
+ [[ -n $PROPOSER_URL ]] && proposer+=(--reflection-base-url $PROPOSER_URL)
31
+ [[ -n $PROPOSER_OPTIONS ]] && proposer+=(--reflection-options $PROPOSER_OPTIONS)
32
+ judge=(--backend ruling --max-workers 1 --base-url $JUDGE --engine $STOCKFISH)
33
+
34
+ deadline=$(( $(date +%s) + HOURS * 3600 ))
35
+ note() { print -r -- "$(date '+%F %T') $*" | tee -a $LOGS/status.txt; }
36
+
37
+ caffeinate -dimsu -w $$ &
38
+ print $$ >$LOGS/driver.pid
39
+ runs=()
40
+ halt() {
41
+ trap - TERM INT
42
+ # Each search stops after its current round and keeps its checkpoint; running this script
43
+ # again resumes them. A second signal ends the searches at once.
44
+ note "stopping after the current round"
45
+ for search in $LOGS/*/search; do [[ -d $search ]] && touch $search/gepa.stop; done
46
+ trap 'pkill -f "^[^ ]*python3 [^ ]*/rulesmith (optimize|evaluate) examples/chess/"' TERM INT
47
+ [[ -n ${stopper:-} ]] && kill $stopper 2>/dev/null
48
+ wait $runs
49
+ rm -f $LOGS/driver.pid
50
+ exit 143
51
+ }
52
+ trap halt TERM INT
53
+
54
+ search() {
55
+ local seed=$1 run=$LOGS/chess-$1
56
+ if [[ -f $run/report.json ]]; then
57
+ note "seed $seed: already done"
58
+ else
59
+ # A search that was interrupted resumes from its checkpoint; one that left none is set aside.
60
+ resume=()
61
+ if [[ -f $run/search/gepa_state.bin ]]; then resume=(--resume)
62
+ elif [[ -d $run ]]; then mv $run $run-failed-$(date +%s)
63
+ fi
64
+ note "seed $seed: searching, $BUDGET games"
65
+ uv run rulesmith optimize $TASK $START $judge --max-metric-calls $BUDGET --seed $seed \
66
+ $proposer $resume --output $run >>$LOGS/chess-$seed.log 2>&1 \
67
+ || { note "seed $seed: search failed, see $LOGS/chess-$seed.log"; return; }
68
+ fi
69
+ if [[ ! -f $run/test.json ]]; then
70
+ note "seed $seed: scoring the winner on test"
71
+ uv run rulesmith evaluate $TASK $run/plan.json $judge --split test --output $run/test.json \
72
+ >>$LOGS/chess-$seed.log 2>&1 || note "seed $seed: test failed, see $LOGS/chess-$seed.log"
73
+ fi
74
+ note "seed $seed: done"
75
+ }
76
+
77
+ # Games take a core each and share nothing, so the seeds run at once.
78
+ for seed in $SEEDS; do search $seed & runs+=($!); done
79
+ ( sleep $(( deadline - $(date +%s) )); note "deadline reached"; kill -TERM $$ ) &
80
+ stopper=$!
81
+ wait $runs
82
+ kill $stopper 2>/dev/null
83
+ rm -f $LOGS/driver.pid
84
+ uv run python3 benchmarks/fitness.py runs/doom runs/chess benchmarks/results/games.json \
85
+ docs/assets/chess-stockfish.svg --tasks chess \
86
+ && note "all finished; chart rebuilt" || note "chart not rebuilt; see above"