rulesmith 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rulesmith-0.1.0/.gitignore +11 -0
- rulesmith-0.1.0/LICENSE +21 -0
- rulesmith-0.1.0/PKG-INFO +131 -0
- rulesmith-0.1.0/README.md +87 -0
- rulesmith-0.1.0/benchmarks/charts.py +312 -0
- rulesmith-0.1.0/benchmarks/chess.sh +86 -0
- rulesmith-0.1.0/benchmarks/doom.sh +104 -0
- rulesmith-0.1.0/benchmarks/escalation.py +117 -0
- rulesmith-0.1.0/benchmarks/escalation.sh +41 -0
- rulesmith-0.1.0/benchmarks/fitness.py +250 -0
- rulesmith-0.1.0/benchmarks/map.py +122 -0
- rulesmith-0.1.0/benchmarks/overnight.sh +166 -0
- rulesmith-0.1.0/benchmarks/results/banking77.json +6078 -0
- rulesmith-0.1.0/benchmarks/results/chess-stockfish-skill0.json +92 -0
- rulesmith-0.1.0/benchmarks/results/clothing.json +8058 -0
- rulesmith-0.1.0/benchmarks/results/escalation.json +5695 -0
- rulesmith-0.1.0/benchmarks/results/games.json +489 -0
- rulesmith-0.1.0/benchmarks/results/instruct/banking77.json +3263 -0
- rulesmith-0.1.0/benchmarks/results/instruct/clothing.json +3241 -0
- rulesmith-0.1.0/benchmarks/results/instruct/jobs.json +3393 -0
- rulesmith-0.1.0/benchmarks/results/instruct/news.json +3733 -0
- rulesmith-0.1.0/benchmarks/results/instruct/trec.json +3050 -0
- rulesmith-0.1.0/benchmarks/results/jobs.json +7935 -0
- rulesmith-0.1.0/benchmarks/results/news.json +7457 -0
- rulesmith-0.1.0/benchmarks/results/training-terms.json +342 -0
- rulesmith-0.1.0/benchmarks/results/trec.json +6807 -0
- rulesmith-0.1.0/benchmarks/results/tuned/clothing.json +3243 -0
- rulesmith-0.1.0/benchmarks/results/tuned/jobs.json +3409 -0
- rulesmith-0.1.0/benchmarks/results/tuned/news.json +3700 -0
- rulesmith-0.1.0/benchmarks/results/tuned/trec.json +3050 -0
- rulesmith-0.1.0/benchmarks/tuned-judge.sh +53 -0
- rulesmith-0.1.0/benchmarks/tuned_judge.py +106 -0
- rulesmith-0.1.0/docs/benchmarks.md +496 -0
- rulesmith-0.1.0/docs/chess.md +258 -0
- rulesmith-0.1.0/docs/cli.md +219 -0
- rulesmith-0.1.0/docs/doom.md +474 -0
- rulesmith-0.1.0/docs/graphs.md +369 -0
- rulesmith-0.1.0/docs/maps.md +196 -0
- rulesmith-0.1.0/docs/optimization.md +192 -0
- rulesmith-0.1.0/docs/python.md +98 -0
- rulesmith-0.1.0/docs/quickstart.md +334 -0
- rulesmith-0.1.0/docs/related-work.md +94 -0
- rulesmith-0.1.0/docs/rules.md +695 -0
- rulesmith-0.1.0/docs/tasks.md +220 -0
- rulesmith-0.1.0/examples/README.md +25 -0
- rulesmith-0.1.0/examples/chess/duel.json +51 -0
- rulesmith-0.1.0/examples/chess/graphs/always-capture.rules +3 -0
- rulesmith-0.1.0/examples/chess/graphs/always-check.rules +1 -0
- rulesmith-0.1.0/examples/chess/graphs/always-quiet.rules +1 -0
- rulesmith-0.1.0/examples/chess/graphs/developer.rules +5 -0
- rulesmith-0.1.0/examples/chess/graphs/greedy.rules +9 -0
- rulesmith-0.1.0/examples/chess/graphs/hunter.rules +14 -0
- rulesmith-0.1.0/examples/chess/graphs/league-best.json +144 -0
- rulesmith-0.1.0/examples/chess/graphs/searched.json +212 -0
- rulesmith-0.1.0/examples/chess/league.json +147 -0
- rulesmith-0.1.0/examples/chess/stockfish.json +80 -0
- rulesmith-0.1.0/examples/doom/basic.json +10 -0
- rulesmith-0.1.0/examples/doom/defend_the_center.json +34 -0
- rulesmith-0.1.0/examples/doom/defend_the_line.json +35 -0
- rulesmith-0.1.0/examples/doom/duel-maps.json +196 -0
- rulesmith-0.1.0/examples/doom/duel.json +192 -0
- rulesmith-0.1.0/examples/doom/e1m1.json +82 -0
- rulesmith-0.1.0/examples/doom/freedoom.json +220 -0
- rulesmith-0.1.0/examples/doom/graphs/duel-attack.json +1 -0
- rulesmith-0.1.0/examples/doom/graphs/duel-hunter.json +782 -0
- rulesmith-0.1.0/examples/doom/graphs/duel-jev.json +702 -0
- rulesmith-0.1.0/examples/doom/graphs/duel-searched.json +471 -0
- rulesmith-0.1.0/examples/doom/graphs/e1m1-exit.json +517 -0
- rulesmith-0.1.0/examples/doom/graphs/e1m1-jev-seed.json +577 -0
- rulesmith-0.1.0/examples/doom/graphs/e1m1-reads.json +267 -0
- rulesmith-0.1.0/examples/doom/graphs/e1m1-searched.json +595 -0
- rulesmith-0.1.0/examples/doom/graphs/e1m1-seed.json +201 -0
- rulesmith-0.1.0/examples/doom/graphs/e1m1-target.json +339 -0
- rulesmith-0.1.0/examples/doom/health_gathering.json +34 -0
- rulesmith-0.1.0/examples/doom/health_gathering_supreme.json +34 -0
- rulesmith-0.1.0/examples/doom/predict_position.json +32 -0
- rulesmith-0.1.0/examples/doom/rocket_basic.json +32 -0
- rulesmith-0.1.0/examples/doom/simpler_basic.json +32 -0
- rulesmith-0.1.0/examples/doom/take_cover.json +32 -0
- rulesmith-0.1.0/examples/order-review/graph.rules +10 -0
- rulesmith-0.1.0/examples/reply-check/graph.rules +7 -0
- rulesmith-0.1.0/examples/support-routing/dag.json +147 -0
- rulesmith-0.1.0/examples/support-routing/evaluation.json +158 -0
- rulesmith-0.1.0/examples/support-routing/graph.rules +5 -0
- rulesmith-0.1.0/examples/support-routing/input.json +3 -0
- rulesmith-0.1.0/examples/support-routing/seed.json +27 -0
- rulesmith-0.1.0/examples/support-routing/task.json +70 -0
- rulesmith-0.1.0/examples/ticket-handling/map.json +47 -0
- rulesmith-0.1.0/examples/ticket-handling/units/billing.rules +5 -0
- rulesmith-0.1.0/examples/ticket-triage/evaluation.json +422 -0
- rulesmith-0.1.0/examples/ticket-triage/graph.rules +6 -0
- rulesmith-0.1.0/examples/ticket-triage/task.json +415 -0
- rulesmith-0.1.0/examples/trust-safety/evaluation.json +7682 -0
- rulesmith-0.1.0/examples/trust-safety/map.json +58 -0
- rulesmith-0.1.0/examples/trust-safety/task.json +1868 -0
- rulesmith-0.1.0/examples/trust-safety/units/intents.rules +1 -0
- rulesmith-0.1.0/examples/trust-safety/units/postings.rules +7 -0
- rulesmith-0.1.0/examples/trust-safety/units/reviews.rules +4 -0
- rulesmith-0.1.0/pyproject.toml +97 -0
- rulesmith-0.1.0/src/rulesmith/__init__.py +1 -0
- rulesmith-0.1.0/src/rulesmith/ablate.py +99 -0
- rulesmith-0.1.0/src/rulesmith/arena.py +172 -0
- rulesmith-0.1.0/src/rulesmith/bench.py +1068 -0
- rulesmith-0.1.0/src/rulesmith/calibrate.py +457 -0
- rulesmith-0.1.0/src/rulesmith/chat_judge.py +205 -0
- rulesmith-0.1.0/src/rulesmith/chess.py +526 -0
- rulesmith-0.1.0/src/rulesmith/clef.py +66 -0
- rulesmith-0.1.0/src/rulesmith/cli.py +1234 -0
- rulesmith-0.1.0/src/rulesmith/diagram.py +226 -0
- rulesmith-0.1.0/src/rulesmith/doom.py +550 -0
- rulesmith-0.1.0/src/rulesmith/extract.py +77 -0
- rulesmith-0.1.0/src/rulesmith/grade.py +85 -0
- rulesmith-0.1.0/src/rulesmith/graph.py +975 -0
- rulesmith-0.1.0/src/rulesmith/label.py +67 -0
- rulesmith-0.1.0/src/rulesmith/level.py +389 -0
- rulesmith-0.1.0/src/rulesmith/maps.py +96 -0
- rulesmith-0.1.0/src/rulesmith/mine.py +313 -0
- rulesmith-0.1.0/src/rulesmith/optimize.py +931 -0
- rulesmith-0.1.0/src/rulesmith/rules.py +1017 -0
- rulesmith-0.1.0/src/rulesmith/runtime.py +711 -0
- rulesmith-0.1.0/src/rulesmith/serve.py +68 -0
- rulesmith-0.1.0/src/rulesmith/tuning.py +134 -0
- rulesmith-0.1.0/tests/conftest.py +212 -0
- rulesmith-0.1.0/tests/test_ablate.py +120 -0
- rulesmith-0.1.0/tests/test_arena.py +394 -0
- rulesmith-0.1.0/tests/test_arithmetic.py +59 -0
- rulesmith-0.1.0/tests/test_bench.py +342 -0
- rulesmith-0.1.0/tests/test_briefing.py +98 -0
- rulesmith-0.1.0/tests/test_calibrate.py +390 -0
- rulesmith-0.1.0/tests/test_charts.py +39 -0
- rulesmith-0.1.0/tests/test_chat_judge.py +235 -0
- rulesmith-0.1.0/tests/test_chess.py +256 -0
- rulesmith-0.1.0/tests/test_clef.py +207 -0
- rulesmith-0.1.0/tests/test_cli.py +749 -0
- rulesmith-0.1.0/tests/test_compose.py +163 -0
- rulesmith-0.1.0/tests/test_conditions.py +242 -0
- rulesmith-0.1.0/tests/test_confidence.py +57 -0
- rulesmith-0.1.0/tests/test_diagram.py +122 -0
- rulesmith-0.1.0/tests/test_doom.py +454 -0
- rulesmith-0.1.0/tests/test_doom_level.py +315 -0
- rulesmith-0.1.0/tests/test_drift.py +86 -0
- rulesmith-0.1.0/tests/test_examples.py +51 -0
- rulesmith-0.1.0/tests/test_extract.py +184 -0
- rulesmith-0.1.0/tests/test_fitness.py +101 -0
- rulesmith-0.1.0/tests/test_given.py +84 -0
- rulesmith-0.1.0/tests/test_grade.py +185 -0
- rulesmith-0.1.0/tests/test_graph.py +376 -0
- rulesmith-0.1.0/tests/test_integration.py +1041 -0
- rulesmith-0.1.0/tests/test_judges.py +320 -0
- rulesmith-0.1.0/tests/test_label.py +70 -0
- rulesmith-0.1.0/tests/test_label_lists.py +151 -0
- rulesmith-0.1.0/tests/test_lazy.py +59 -0
- rulesmith-0.1.0/tests/test_length.py +50 -0
- rulesmith-0.1.0/tests/test_lists.py +110 -0
- rulesmith-0.1.0/tests/test_live.py +59 -0
- rulesmith-0.1.0/tests/test_map.py +300 -0
- rulesmith-0.1.0/tests/test_map_demo.py +42 -0
- rulesmith-0.1.0/tests/test_measured.py +180 -0
- rulesmith-0.1.0/tests/test_mine.py +226 -0
- rulesmith-0.1.0/tests/test_openai_judge.py +64 -0
- rulesmith-0.1.0/tests/test_outputs.py +62 -0
- rulesmith-0.1.0/tests/test_proposer.py +60 -0
- rulesmith-0.1.0/tests/test_resume.py +142 -0
- rulesmith-0.1.0/tests/test_round_trip.py +146 -0
- rulesmith-0.1.0/tests/test_rules.py +381 -0
- rulesmith-0.1.0/tests/test_ruling.py +188 -0
- rulesmith-0.1.0/tests/test_runtime.py +317 -0
- rulesmith-0.1.0/tests/test_scoring.py +185 -0
- rulesmith-0.1.0/tests/test_serve.py +91 -0
- rulesmith-0.1.0/tests/test_tuning.py +194 -0
- rulesmith-0.1.0/tests/test_unit_search.py +222 -0
- rulesmith-0.1.0/tests/test_vote.py +119 -0
- rulesmith-0.1.0/uv.lock +2867 -0
rulesmith-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 rulesmith contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in
|
|
13
|
+
all copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
|
21
|
+
THE SOFTWARE.
|
rulesmith-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: rulesmith
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Search for rule programs that ask a small judge for typed decisions, with DSPy and GEPA.
|
|
5
|
+
Project-URL: Homepage, https://github.com/bradAGI/rulesmith
|
|
6
|
+
Project-URL: Repository, https://github.com/bradAGI/rulesmith
|
|
7
|
+
Project-URL: Issues, https://github.com/bradAGI/rulesmith/issues
|
|
8
|
+
Author: rulesmith contributors
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: decision graphs,dspy,gepa,llm,program search,rules
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Requires-Python: <3.15,>=3.12
|
|
21
|
+
Requires-Dist: dspy==3.4.0
|
|
22
|
+
Requires-Dist: gepa==0.1.4
|
|
23
|
+
Requires-Dist: httpx<1,>=0.28
|
|
24
|
+
Requires-Dist: pydantic<3,>=2.11
|
|
25
|
+
Requires-Dist: typesafe-sdk<0.7,>=0.6
|
|
26
|
+
Provides-Extra: bench
|
|
27
|
+
Requires-Dist: huggingface-hub<2,>=1; extra == 'bench'
|
|
28
|
+
Requires-Dist: pyarrow<24,>=17; extra == 'bench'
|
|
29
|
+
Requires-Dist: scikit-learn<2,>=1.5; extra == 'bench'
|
|
30
|
+
Provides-Extra: chess
|
|
31
|
+
Requires-Dist: chess<2,>=1.11; extra == 'chess'
|
|
32
|
+
Provides-Extra: clef
|
|
33
|
+
Requires-Dist: huggingface-hub<2,>=1; extra == 'clef'
|
|
34
|
+
Requires-Dist: pillow<13,>=11; extra == 'clef'
|
|
35
|
+
Requires-Dist: torch<3,>=2.11; extra == 'clef'
|
|
36
|
+
Requires-Dist: torchvision<1,>=0.26; extra == 'clef'
|
|
37
|
+
Requires-Dist: transformers[torch]<6,>=5.10.2; extra == 'clef'
|
|
38
|
+
Provides-Extra: doom
|
|
39
|
+
Requires-Dist: imageio-ffmpeg<1,>=0.6; extra == 'doom'
|
|
40
|
+
Requires-Dist: numpy<3,>=2; extra == 'doom'
|
|
41
|
+
Requires-Dist: pillow<13,>=11; extra == 'doom'
|
|
42
|
+
Requires-Dist: vizdoom<2,>=1.3; extra == 'doom'
|
|
43
|
+
Description-Content-Type: text/markdown
|
|
44
|
+
|
|
45
|
+
# rulesmith
|
|
46
|
+
|
|
47
|
+

|
|
48
|
+
|
|
49
|
+
**rulesmith lets DSPy build decision graphs for you.** Give it a classification
|
|
50
|
+
task and labeled examples. It searches for a small program of rules: cheap tests
|
|
51
|
+
on the input's fields settle the cases they can, and a local or hosted judge is
|
|
52
|
+
asked typed questions about the rest. [GEPA](https://github.com/gepa-ai/gepa)
|
|
53
|
+
keeps the revisions that score better, and the result is a rules file you can
|
|
54
|
+
read, edit, run, and serve.
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
read has_company_logo in [0, 1]
|
|
58
|
+
read length of company_profile in [0, 10000]
|
|
59
|
+
when has_company_logo == 0 and length of company_profile == 0: fraudulent
|
|
60
|
+
always: ask "Is this job posting a scam?" [fraudulent: "fraudulent"] [genuine: "genuine"]
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+

|
|
64
|
+
|
|
65
|
+
The [benchmarks](https://github.com/bradAGI/rulesmith/blob/main/docs/benchmarks.md) have every dataset, including where rulesmith
|
|
66
|
+
does not help: on one-sentence inputs it ties DSPy, and on job postings a TF-IDF
|
|
67
|
+
classifier still beats it.
|
|
68
|
+
|
|
69
|
+
An independent research prototype, not affiliated with TypeSafe.
|
|
70
|
+
|
|
71
|
+
## Installation
|
|
72
|
+
|
|
73
|
+
```sh
|
|
74
|
+
pip install rulesmith
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
Python 3.12 to 3.14. Extras for the games and benchmarks are listed in the
|
|
78
|
+
[quickstart](https://github.com/bradAGI/rulesmith/blob/main/docs/quickstart.md#installation).
|
|
79
|
+
|
|
80
|
+
## Quickstart
|
|
81
|
+
|
|
82
|
+
With a judge running ([Ruling](https://github.com/bradAGI/ruling) on a Mac,
|
|
83
|
+
Cloudflare's Clef-flash on an NVIDIA GPU or a Mac, TypeSafe's hosted Jev, or any
|
|
84
|
+
OpenAI-compatible chat API that returns log-probabilities), any OpenAI-compatible model to write the rules, and a
|
|
85
|
+
labeled task such as [this one](https://github.com/bradAGI/rulesmith/blob/main/examples/support-routing/task.json):
|
|
86
|
+
|
|
87
|
+
```sh
|
|
88
|
+
export OPENAI_API_KEY=local-no-auth # a local model server takes any key
|
|
89
|
+
rulesmith optimize task.json --backend ruling --max-workers 1 \
|
|
90
|
+
--reflection-model "openai/<model>" --reflection-base-url http://127.0.0.1:8080/v1 \
|
|
91
|
+
--output runs/my-task
|
|
92
|
+
rulesmith run runs/my-task/plan.rules input.json --backend ruling
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
The [quickstart](https://github.com/bradAGI/rulesmith/blob/main/docs/quickstart.md) walks through it with every command's output.
|
|
96
|
+
|
|
97
|
+
## Beyond classification
|
|
98
|
+
|
|
99
|
+
The same search plays games. Pitted against a league of other graphs, this one learned to
|
|
100
|
+
hunt in Doom deathmatch and beat the champion 12-0, with no model calls at all
|
|
101
|
+
([how](https://github.com/bradAGI/rulesmith/blob/main/docs/doom.md)):
|
|
102
|
+
|
|
103
|
+

|
|
104
|
+
|
|
105
|
+
## Learn more
|
|
106
|
+
|
|
107
|
+
- [Quickstart](https://github.com/bradAGI/rulesmith/blob/main/docs/quickstart.md): installation, judges, and a task to a running graph.
|
|
108
|
+
- [Notebook](https://github.com/bradAGI/rulesmith/blob/main/notebooks/quickstart.ipynb): rules for spotting fake job postings, end to end in Colab with an OpenAI key.
|
|
109
|
+
- [Local notebook](https://github.com/bradAGI/rulesmith/blob/main/notebooks/local.ipynb): the same with Clef-flash judging on your Mac or NVIDIA GPU, free per call.
|
|
110
|
+
- [Tasks](https://github.com/bradAGI/rulesmith/blob/main/docs/tasks.md): the task file, drafting labels, structured inputs, costs, and graders.
|
|
111
|
+
- [Rules](https://github.com/bradAGI/rulesmith/blob/main/docs/rules.md): the language search writes, confidence, calibration, escalation, and extraction.
|
|
112
|
+
- [Graph format](https://github.com/bradAGI/rulesmith/blob/main/docs/graphs.md): the JSON behind each node, execution, and drawing a graph.
|
|
113
|
+
- [Maps](https://github.com/bradAGI/rulesmith/blob/main/docs/maps.md): many decisions as one map, searched a unit at a time.
|
|
114
|
+
- [Optimization](https://github.com/bradAGI/rulesmith/blob/main/docs/optimization.md): how search runs, its settings, and its limits.
|
|
115
|
+
- [Command line](https://github.com/bradAGI/rulesmith/blob/main/docs/cli.md): every command, including `serve` and `drift`, and the backends.
|
|
116
|
+
- [Python](https://github.com/bradAGI/rulesmith/blob/main/docs/python.md): running and building graphs from code.
|
|
117
|
+
- [Benchmarks](https://github.com/bradAGI/rulesmith/blob/main/docs/benchmarks.md): rulesmith against DSPy, TF-IDF and direct calls on five public datasets.
|
|
118
|
+
- [Related work](https://github.com/bradAGI/rulesmith/blob/main/docs/related-work.md): when to use rulesmith and when DSPy, and what came before.
|
|
119
|
+
- [Doom](https://github.com/bradAGI/rulesmith/blob/main/docs/doom.md) and [Chess](https://github.com/bradAGI/rulesmith/blob/main/docs/chess.md): graphs that play.
|
|
120
|
+
- [Claude Code skill](https://github.com/bradAGI/rulesmith/blob/main/skills/rulesmith/SKILL.md): copy `skills/rulesmith` into `~/.claude/skills/`.
|
|
121
|
+
|
|
122
|
+
## Development
|
|
123
|
+
|
|
124
|
+
```sh
|
|
125
|
+
uv sync --locked
|
|
126
|
+
uv run pytest -q
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
## License
|
|
130
|
+
|
|
131
|
+
[MIT](https://github.com/bradAGI/rulesmith/blob/main/LICENSE).
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
# rulesmith
|
|
2
|
+
|
|
3
|
+

|
|
4
|
+
|
|
5
|
+
**rulesmith lets DSPy build decision graphs for you.** Give it a classification
|
|
6
|
+
task and labeled examples. It searches for a small program of rules: cheap tests
|
|
7
|
+
on the input's fields settle the cases they can, and a local or hosted judge is
|
|
8
|
+
asked typed questions about the rest. [GEPA](https://github.com/gepa-ai/gepa)
|
|
9
|
+
keeps the revisions that score better, and the result is a rules file you can
|
|
10
|
+
read, edit, run, and serve.
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
read has_company_logo in [0, 1]
|
|
14
|
+
read length of company_profile in [0, 10000]
|
|
15
|
+
when has_company_logo == 0 and length of company_profile == 0: fraudulent
|
|
16
|
+
always: ask "Is this job posting a scam?" [fraudulent: "fraudulent"] [genuine: "genuine"]
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+

|
|
20
|
+
|
|
21
|
+
The [benchmarks](docs/benchmarks.md) have every dataset, including where rulesmith
|
|
22
|
+
does not help: on one-sentence inputs it ties DSPy, and on job postings a TF-IDF
|
|
23
|
+
classifier still beats it.
|
|
24
|
+
|
|
25
|
+
An independent research prototype, not affiliated with TypeSafe.
|
|
26
|
+
|
|
27
|
+
## Installation
|
|
28
|
+
|
|
29
|
+
```sh
|
|
30
|
+
pip install rulesmith
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
Python 3.12 to 3.14. Extras for the games and benchmarks are listed in the
|
|
34
|
+
[quickstart](docs/quickstart.md#installation).
|
|
35
|
+
|
|
36
|
+
## Quickstart
|
|
37
|
+
|
|
38
|
+
With a judge running ([Ruling](https://github.com/bradAGI/ruling) on a Mac,
|
|
39
|
+
Cloudflare's Clef-flash on an NVIDIA GPU or a Mac, TypeSafe's hosted Jev, or any
|
|
40
|
+
OpenAI-compatible chat API that returns log-probabilities), any OpenAI-compatible model to write the rules, and a
|
|
41
|
+
labeled task such as [this one](examples/support-routing/task.json):
|
|
42
|
+
|
|
43
|
+
```sh
|
|
44
|
+
export OPENAI_API_KEY=local-no-auth # a local model server takes any key
|
|
45
|
+
rulesmith optimize task.json --backend ruling --max-workers 1 \
|
|
46
|
+
--reflection-model "openai/<model>" --reflection-base-url http://127.0.0.1:8080/v1 \
|
|
47
|
+
--output runs/my-task
|
|
48
|
+
rulesmith run runs/my-task/plan.rules input.json --backend ruling
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
The [quickstart](docs/quickstart.md) walks through it with every command's output.
|
|
52
|
+
|
|
53
|
+
## Beyond classification
|
|
54
|
+
|
|
55
|
+
The same search plays games. Pitted against a league of other graphs, this one learned to
|
|
56
|
+
hunt in Doom deathmatch and beat the champion 12-0, with no model calls at all
|
|
57
|
+
([how](docs/doom.md)):
|
|
58
|
+
|
|
59
|
+

|
|
60
|
+
|
|
61
|
+
## Learn more
|
|
62
|
+
|
|
63
|
+
- [Quickstart](docs/quickstart.md): installation, judges, and a task to a running graph.
|
|
64
|
+
- [Notebook](notebooks/quickstart.ipynb): rules for spotting fake job postings, end to end in Colab with an OpenAI key.
|
|
65
|
+
- [Local notebook](notebooks/local.ipynb): the same with Clef-flash judging on your Mac or NVIDIA GPU, free per call.
|
|
66
|
+
- [Tasks](docs/tasks.md): the task file, drafting labels, structured inputs, costs, and graders.
|
|
67
|
+
- [Rules](docs/rules.md): the language search writes, confidence, calibration, escalation, and extraction.
|
|
68
|
+
- [Graph format](docs/graphs.md): the JSON behind each node, execution, and drawing a graph.
|
|
69
|
+
- [Maps](docs/maps.md): many decisions as one map, searched a unit at a time.
|
|
70
|
+
- [Optimization](docs/optimization.md): how search runs, its settings, and its limits.
|
|
71
|
+
- [Command line](docs/cli.md): every command, including `serve` and `drift`, and the backends.
|
|
72
|
+
- [Python](docs/python.md): running and building graphs from code.
|
|
73
|
+
- [Benchmarks](docs/benchmarks.md): rulesmith against DSPy, TF-IDF and direct calls on five public datasets.
|
|
74
|
+
- [Related work](docs/related-work.md): when to use rulesmith and when DSPy, and what came before.
|
|
75
|
+
- [Doom](docs/doom.md) and [Chess](docs/chess.md): graphs that play.
|
|
76
|
+
- [Claude Code skill](skills/rulesmith/SKILL.md): copy `skills/rulesmith` into `~/.claude/skills/`.
|
|
77
|
+
|
|
78
|
+
## Development
|
|
79
|
+
|
|
80
|
+
```sh
|
|
81
|
+
uv sync --locked
|
|
82
|
+
uv run pytest -q
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
## License
|
|
86
|
+
|
|
87
|
+
[MIT](LICENSE).
|
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
"""The benchmark charts in the docs, drawn from the results files.
|
|
2
|
+
|
|
3
|
+
python3 benchmarks/charts.py
|
|
4
|
+
|
|
5
|
+
Writes docs/assets/benchmarks.svg, accuracy against judge calls per row for rulesmith's search
|
|
6
|
+
and the baselines on each dataset; docs/assets/escalation.svg, accuracy against the share of
|
|
7
|
+
rows sent to the larger judge; and docs/assets/claim.svg, the README's headline: search against
|
|
8
|
+
DSPy's typed classifier on the same tuned judge. Every number is a mean over the seeds in
|
|
9
|
+
benchmarks/results; the tables in docs/benchmarks.md carry the spread and the tests of
|
|
10
|
+
significance.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from statistics import mean
|
|
16
|
+
|
|
17
|
+
ROOT = Path(__file__).resolve().parents[1]
|
|
18
|
+
RESULTS = ROOT / "benchmarks" / "results"
|
|
19
|
+
ASSETS = ROOT / "docs" / "assets"
|
|
20
|
+
|
|
21
|
+
# Where rules over fields help first, then where the input is one piece of free text.
|
|
22
|
+
DATASETS = ("clothing", "jobs", "news", "trec", "banking77")
|
|
23
|
+
# The baselines that ask a language model on every row; the best of them per dataset is the
|
|
24
|
+
# fair comparison, so search is never shown beside a weak one.
|
|
25
|
+
ASKING = {
|
|
26
|
+
"direct": "direct call",
|
|
27
|
+
"dspy": "DSPy",
|
|
28
|
+
"dspy-cot": "DSPy, reasoning",
|
|
29
|
+
"dspy-typed": "DSPy typed",
|
|
30
|
+
}
|
|
31
|
+
# The README's headline: where the input has fields, on the judge tuned on each seed's examples.
|
|
32
|
+
CLAIM = (
|
|
33
|
+
("jobs", "Job postings: scam or genuine?"),
|
|
34
|
+
("clothing", "Clothing reviews: recommend or not?"),
|
|
35
|
+
)
|
|
36
|
+
ESCALATION = (("small", "4B alone"), ("large", "Clef-flash alone"), ("escalate", "escalating"))
|
|
37
|
+
|
|
38
|
+
# rulesmith is the one hue; every baseline is context, in the neutral, and named beside its mark.
|
|
39
|
+
STYLE = """
|
|
40
|
+
.bg { fill: #fcfcfb } .ink { fill: #0b0b0b } .muted { fill: #52514e }
|
|
41
|
+
.grid { stroke: #e4e3df } .axis { stroke: #b5b4ae }
|
|
42
|
+
.us { fill: #2a78d6 } .them { fill: #6b6a66 } .ring { stroke: #fcfcfb }
|
|
43
|
+
.mix { stroke: #6b6a66 }
|
|
44
|
+
.halo { stroke: #fcfcfb; stroke-width: 4px; paint-order: stroke; stroke-linejoin: round }
|
|
45
|
+
@media (prefers-color-scheme: dark) {
|
|
46
|
+
.bg { fill: #1a1a19 } .ink { fill: #ffffff } .muted { fill: #c3c2b7 }
|
|
47
|
+
.grid { stroke: #33322f } .axis { stroke: #57564f }
|
|
48
|
+
.us { fill: #3987e5 } .them { fill: #8f8e89 } .ring { stroke: #1a1a19 }
|
|
49
|
+
.mix { stroke: #8f8e89 }
|
|
50
|
+
.halo { stroke: #1a1a19 }
|
|
51
|
+
}
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
PANEL, WIDE, GAP, LEFT, TOP, PLOT = 176, 300, 62, 52, 74, 190
|
|
55
|
+
FONT = 'font-family="system-ui, sans-serif" font-size="12"'
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def by_method(dataset: str) -> dict[str, tuple[float, float]]:
|
|
59
|
+
"""Mean accuracy and judge calls per row of each method on a dataset, over its seeds."""
|
|
60
|
+
runs = json.loads((RESULTS / f"{dataset}.json").read_text())["runs"]
|
|
61
|
+
methods = {r["method"] for r in runs}
|
|
62
|
+
return {
|
|
63
|
+
m: (
|
|
64
|
+
mean(r["accuracy"] for r in runs if r["method"] == m),
|
|
65
|
+
mean(r["calls_per_example"] for r in runs if r["method"] == m),
|
|
66
|
+
)
|
|
67
|
+
for m in methods
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def comparison(found: dict[str, tuple[float, float]]) -> list[tuple[str, float, float, bool]]:
|
|
72
|
+
"""What a panel shows: search, the best baseline that asks the model on every row, and the
|
|
73
|
+
trained classifier that never asks, as (name, calls, accuracy, is rulesmith)."""
|
|
74
|
+
best = max((m for m in ASKING if m in found), key=lambda m: found[m][0])
|
|
75
|
+
return [
|
|
76
|
+
("TF-IDF", found["tfidf"][1], found["tfidf"][0], False),
|
|
77
|
+
(ASKING[best], found[best][1], found[best][0], False),
|
|
78
|
+
("rulesmith", found["search"][1], found["search"][0], True),
|
|
79
|
+
]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def escalated() -> dict[str, list[tuple[str, float, float]]]:
|
|
83
|
+
"""Per dataset, each way of judging as (name, larger-judge calls per row, accuracy)."""
|
|
84
|
+
runs = json.loads((RESULTS / "escalation.json").read_text())["runs"]
|
|
85
|
+
found = {}
|
|
86
|
+
for dataset in sorted({r["dataset"] for r in runs}):
|
|
87
|
+
found[dataset] = []
|
|
88
|
+
for graph, name in ESCALATION:
|
|
89
|
+
mine = [r for r in runs if r["dataset"] == dataset and r["graph"] == graph]
|
|
90
|
+
large = mean(r["calls_per_row"]["large"] for r in mine)
|
|
91
|
+
found[dataset].append((name, large, mean(r["accuracy"] for r in mine)))
|
|
92
|
+
return found
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def tuned(dataset: str, method: str) -> list[dict]:
|
|
96
|
+
"""A method's runs on the tuned judge, seed by seed, so pooled rows line up across methods."""
|
|
97
|
+
runs = json.loads((RESULTS / "tuned" / f"{dataset}.json").read_text())["runs"]
|
|
98
|
+
return sorted((r for r in runs if r["method"] == method), key=lambda r: r["seed"])
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def p_value(p: float) -> str:
|
|
102
|
+
"""A p-value as a reader writes it: 0.66, or 3×10⁻⁶ rather than 3.2e-06."""
|
|
103
|
+
if p >= 0.001:
|
|
104
|
+
return f"{p:.2g}"
|
|
105
|
+
mantissa, exponent = f"{p:.0e}".split("e")
|
|
106
|
+
raised = str(int(exponent)).translate(str.maketrans("-0123456789", "⁻⁰¹²³⁴⁵⁶⁷⁸⁹"))
|
|
107
|
+
return f"{mantissa}×10{raised}"
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def ticks(low: float, high: float) -> tuple[float, float, list[float]]:
|
|
111
|
+
"""A y range padded to the nearest 0.05 around the values, and its gridlines."""
|
|
112
|
+
step = 0.05 if high - low < 0.25 else 0.1
|
|
113
|
+
bottom = max(0.0, (int(low / step) - (0 if low % step else 1)) * step)
|
|
114
|
+
top = min(1.0, (int(high / step) + 1) * step)
|
|
115
|
+
count = round((top - bottom) / step)
|
|
116
|
+
return bottom, top, [round(bottom + i * step, 2) for i in range(count + 1)]
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def labels(points: list[tuple[float, float]]) -> list[int]:
|
|
120
|
+
"""Which way each point's label goes: up, unless a higher point sits near enough that the
|
|
121
|
+
labels would meet, in which case the lower one is labeled below."""
|
|
122
|
+
sides = [-1] * len(points)
|
|
123
|
+
for i, (x, y) in enumerate(points):
|
|
124
|
+
for j, (x2, y2) in enumerate(points):
|
|
125
|
+
if i != j and abs(x - x2) < 70 and 0 <= y - y2 < 26:
|
|
126
|
+
sides[i] = 1
|
|
127
|
+
return sides
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def panel(left: float, title: str, points, x_label: str, segment=None, size=PANEL) -> list[str]:
|
|
131
|
+
"""One small multiple: its own y scale, x from 0 to 1, each point named beside it."""
|
|
132
|
+
values = [y for _, _, y, *_ in points]
|
|
133
|
+
bottom, top, grid = ticks(min(values), max(values))
|
|
134
|
+
x = lambda v: left + 10 + v * (size - 20) # noqa: E731
|
|
135
|
+
y = lambda v: TOP + PLOT - (v - bottom) / (top - bottom) * PLOT # noqa: E731
|
|
136
|
+
out = [f'<text x="{left}" y="{TOP - 14}" font-weight="600" class="ink">{title}</text>']
|
|
137
|
+
for g in grid:
|
|
138
|
+
out.append(
|
|
139
|
+
f'<line x1="{left}" y1="{y(g):.1f}" x2="{left + size}" y2="{y(g):.1f}" class="grid"/>'
|
|
140
|
+
)
|
|
141
|
+
out.append(
|
|
142
|
+
f'<text x="{left - 6}" y="{y(g) + 4:.1f}" text-anchor="end" class="muted">'
|
|
143
|
+
f"{g:.2f}</text>"
|
|
144
|
+
)
|
|
145
|
+
base = TOP + PLOT
|
|
146
|
+
out.append(f'<line x1="{left}" y1="{base}" x2="{left + size}" y2="{base}" class="axis"/>')
|
|
147
|
+
for v in (0, 0.5, 1):
|
|
148
|
+
out.append(
|
|
149
|
+
f'<text x="{x(v):.1f}" y="{base + 16}" text-anchor="middle" class="muted">{v:g}</text>'
|
|
150
|
+
)
|
|
151
|
+
out.append(
|
|
152
|
+
f'<text x="{left + size / 2}" y="{base + 34}" text-anchor="middle" class="muted">'
|
|
153
|
+
f"{x_label}</text>"
|
|
154
|
+
)
|
|
155
|
+
if segment:
|
|
156
|
+
(x1, y1), (x2, y2) = segment
|
|
157
|
+
out.append(
|
|
158
|
+
f'<line x1="{x(x1):.1f}" y1="{y(y1):.1f}" x2="{x(x2):.1f}" y2="{y(y2):.1f}" '
|
|
159
|
+
'class="mix" stroke-width="1.5" stroke-dasharray="4 4"/>'
|
|
160
|
+
)
|
|
161
|
+
placed = [(x(c), y(a)) for _, c, a, *_ in points]
|
|
162
|
+
for (name, calls, accuracy, *ours), (px, py), side in zip(
|
|
163
|
+
points, placed, labels(placed), strict=True
|
|
164
|
+
):
|
|
165
|
+
kind = "us" if ours and ours[0] else "them"
|
|
166
|
+
anchor = "start" if px < left + 30 else "end" if px > left + size - 30 else "middle"
|
|
167
|
+
weight = ' font-weight="600"' if kind == "us" else ""
|
|
168
|
+
out.append(
|
|
169
|
+
f'<circle cx="{px:.1f}" cy="{py:.1f}" r="5" class="{kind} ring" stroke-width="2">'
|
|
170
|
+
f"<title>{name}: {accuracy:.3f} accuracy, {calls:.2f} calls per row</title></circle>"
|
|
171
|
+
)
|
|
172
|
+
ly = py - 10 if side < 0 else py + 19
|
|
173
|
+
out.append(
|
|
174
|
+
f'<text x="{px:.1f}" y="{ly:.1f}" text-anchor="{anchor}" class="ink halo"{weight}>'
|
|
175
|
+
f"{name} {accuracy:.3f}</text>"
|
|
176
|
+
)
|
|
177
|
+
return out
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def svg(width: int, height: int, title: str, subtitle: str, body: list[str]) -> str:
|
|
181
|
+
return "\n".join(
|
|
182
|
+
[
|
|
183
|
+
f'<svg xmlns="http://www.w3.org/2000/svg" width="{width}" height="{height}" '
|
|
184
|
+
f'viewBox="0 0 {width} {height}" {FONT}>',
|
|
185
|
+
f"<style>{STYLE}</style>",
|
|
186
|
+
f'<rect width="{width}" height="{height}" class="bg"/>',
|
|
187
|
+
f'<text x="16" y="24" font-size="14" font-weight="600" class="ink">{title}</text>',
|
|
188
|
+
f'<text x="16" y="42" class="muted">{subtitle}</text>',
|
|
189
|
+
*body,
|
|
190
|
+
"</svg>",
|
|
191
|
+
"",
|
|
192
|
+
]
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def benchmarks_chart() -> str:
|
|
197
|
+
body = []
|
|
198
|
+
for i, dataset in enumerate(DATASETS):
|
|
199
|
+
shown = comparison(by_method(dataset))
|
|
200
|
+
body += panel(LEFT + i * (PANEL + GAP), dataset, shown, "judge calls per row")
|
|
201
|
+
width = LEFT + len(DATASETS) * (PANEL + GAP) - GAP + 24
|
|
202
|
+
return svg(
|
|
203
|
+
width,
|
|
204
|
+
TOP + PLOT + 52,
|
|
205
|
+
"Accuracy against judge calls per row, on 300 held-out rows",
|
|
206
|
+
"rulesmith's searched rules against the best baseline that asks on every row, and a "
|
|
207
|
+
"classifier that never asks. Means of three seeds; each panel has its own scale.",
|
|
208
|
+
body,
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def escalation_chart() -> str:
|
|
213
|
+
body = []
|
|
214
|
+
for i, (dataset, points) in enumerate(escalated().items()):
|
|
215
|
+
marked = [(name, calls, acc, name == "escalating") for name, calls, acc in points]
|
|
216
|
+
ends = [(c, a) for n, c, a in points if n != "escalating"]
|
|
217
|
+
body += panel(
|
|
218
|
+
LEFT + i * (WIDE + GAP), dataset, marked, "larger-judge calls per row", ends, WIDE
|
|
219
|
+
)
|
|
220
|
+
width = LEFT + 2 * (WIDE + GAP) - GAP + 24
|
|
221
|
+
return svg(
|
|
222
|
+
width,
|
|
223
|
+
TOP + PLOT + 52,
|
|
224
|
+
"Escalating to a larger judge only where the small one is unsure",
|
|
225
|
+
"Above the dashed line is better than sending that share of rows to the larger judge at "
|
|
226
|
+
"random. Means of three seeds.",
|
|
227
|
+
body,
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def bar(x: float, y: float, length: float, value: str, kind: str) -> list[str]:
|
|
232
|
+
"""A bar from zero, and its value at the end of it in the text color."""
|
|
233
|
+
weight = ' font-weight="600"' if kind == "us" else ""
|
|
234
|
+
return [
|
|
235
|
+
f'<rect x="{x}" y="{y + 2}" width="{length:.1f}" height="18" rx="2" class="{kind}"/>',
|
|
236
|
+
f'<text x="{x + length + 6:.1f}" y="{y + 15}" class="ink"{weight}>{value}</text>',
|
|
237
|
+
]
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def claim_chart() -> str:
|
|
241
|
+
from rulesmith.bench import mcnemar
|
|
242
|
+
|
|
243
|
+
width, left, accuracy_width, calls_left, calls_width = 820, 200, 360, 640, 110
|
|
244
|
+
body = [
|
|
245
|
+
f'<text x="{left}" y="{TOP}" class="muted" font-weight="600">accuracy</text>',
|
|
246
|
+
f'<text x="{calls_left}" y="{TOP}" class="muted" font-weight="600">'
|
|
247
|
+
"model calls per row</text>",
|
|
248
|
+
]
|
|
249
|
+
y = TOP + 34
|
|
250
|
+
for dataset, title in CLAIM:
|
|
251
|
+
ours, theirs = tuned(dataset, "search"), tuned(dataset, "dspy-typed")
|
|
252
|
+
pooled = [[c for r in runs for c in r["correct"]] for runs in (ours, theirs)]
|
|
253
|
+
p = mcnemar(*pooled)
|
|
254
|
+
rows = [
|
|
255
|
+
(
|
|
256
|
+
"rulesmith",
|
|
257
|
+
mean(r["accuracy"] for r in ours),
|
|
258
|
+
mean(r["calls_per_example"] for r in ours),
|
|
259
|
+
"us",
|
|
260
|
+
),
|
|
261
|
+
(
|
|
262
|
+
"DSPy typed classifier",
|
|
263
|
+
mean(r["accuracy"] for r in theirs),
|
|
264
|
+
mean(r["calls_per_example"] for r in theirs),
|
|
265
|
+
"them",
|
|
266
|
+
),
|
|
267
|
+
]
|
|
268
|
+
gain = (rows[0][1] - rows[1][1]) * 100
|
|
269
|
+
verdict = f"+{gain:.1f} points" if p < 0.05 and gain > 0 else "tie"
|
|
270
|
+
share = rows[0][2] / rows[1][2]
|
|
271
|
+
body.append(
|
|
272
|
+
f'<text x="16" y="{y}" class="ink" font-size="14" font-weight="600">{title}</text>'
|
|
273
|
+
f'<text x="{left + 230}" y="{y}" class="muted">{verdict} · p = {p_value(p)} · '
|
|
274
|
+
f"{share:.0%} of the calls</text>"
|
|
275
|
+
)
|
|
276
|
+
y += 14
|
|
277
|
+
for name, accuracy, calls, kind in rows:
|
|
278
|
+
weight = ' font-weight="600"' if kind == "us" else ""
|
|
279
|
+
body += [
|
|
280
|
+
f'<text x="{left - 10}" y="{y + 15}" text-anchor="end" class="ink"{weight}>'
|
|
281
|
+
f"{name}</text>",
|
|
282
|
+
*bar(left, y, accuracy * accuracy_width, f"{accuracy:.3f}", kind),
|
|
283
|
+
*bar(calls_left, y, calls * calls_width, f"{calls:.2f}", kind),
|
|
284
|
+
]
|
|
285
|
+
y += 26
|
|
286
|
+
y += 30
|
|
287
|
+
body.append(
|
|
288
|
+
f'<text x="16" y="{y}" class="muted">Means of 3 seeds, 300 held-out rows each; p from '
|
|
289
|
+
"exact McNemar tests over the pooled rows. Bars start at zero.</text>"
|
|
290
|
+
)
|
|
291
|
+
body.append(
|
|
292
|
+
f'<text x="16" y="{y + 18}" class="muted">On one-sentence inputs (TREC, Banking77) '
|
|
293
|
+
"rulesmith ties DSPy; on job postings a TF-IDF classifier scores higher still.</text>"
|
|
294
|
+
)
|
|
295
|
+
return svg(
|
|
296
|
+
width,
|
|
297
|
+
y + 36,
|
|
298
|
+
"Same 4B judge, same 100 examples, a fraction of the calls",
|
|
299
|
+
"rulesmith against DSPy's typed classifier (GEPA + ReAnchor), the judge fine-tuned on each "
|
|
300
|
+
"seed's examples.",
|
|
301
|
+
body,
|
|
302
|
+
)
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def main() -> None:
|
|
306
|
+
(ASSETS / "benchmarks.svg").write_text(benchmarks_chart())
|
|
307
|
+
(ASSETS / "escalation.svg").write_text(escalation_chart())
|
|
308
|
+
(ASSETS / "claim.svg").write_text(claim_chart())
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
if __name__ == "__main__":
|
|
312
|
+
main()
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
#!/bin/zsh
|
|
2
|
+
# The chess searches against Stockfish behind docs/assets/fitness.svg: from the greedy graph,
|
|
3
|
+
# several search seeds at once, each with the same budget of games. A finished search is kept,
|
|
4
|
+
# so running this again picks up where an interrupted run stopped.
|
|
5
|
+
#
|
|
6
|
+
# benchmarks/chess.sh # three seeds, 1000 games each
|
|
7
|
+
# BUDGET=200 SEEDS="0" benchmarks/chess.sh
|
|
8
|
+
# kill $(cat runs/chess/driver.pid) # stop it, and only it
|
|
9
|
+
#
|
|
10
|
+
# Needs: Stockfish at $STOCKFISH, a proposer (a hosted model with its key in the environment,
|
|
11
|
+
# or a local server at $PROPOSER_URL), and Ruling at $JUDGE for any graph that asks it.
|
|
12
|
+
set -u
|
|
13
|
+
cd "${0:A:h}/.."
|
|
14
|
+
|
|
15
|
+
HOURS=${HOURS:-6}
|
|
16
|
+
BUDGET=${BUDGET:-1000}
|
|
17
|
+
SEEDS=(${=SEEDS:-0 1 2})
|
|
18
|
+
STOCKFISH=${STOCKFISH:-$HOME/.local/bin/stockfish}
|
|
19
|
+
JUDGE=${JUDGE:-http://127.0.0.1:8010}
|
|
20
|
+
PROPOSER_MODEL=${PROPOSER_MODEL:-openrouter/qwen/qwen3.8-27b}
|
|
21
|
+
PROPOSER_URL=${PROPOSER_URL-}
|
|
22
|
+
PROPOSER_OPTIONS=${PROPOSER_OPTIONS-'{"reasoning": {"max_tokens": 2048}}'}
|
|
23
|
+
TASK=examples/chess/stockfish.json
|
|
24
|
+
START=examples/chess/graphs/greedy.rules
|
|
25
|
+
LOGS=runs/chess
|
|
26
|
+
mkdir -p $LOGS benchmarks/results
|
|
27
|
+
export OPENAI_API_KEY=${OPENAI_API_KEY:-local-no-auth}
|
|
28
|
+
|
|
29
|
+
proposer=(--reflection-model $PROPOSER_MODEL --reflection-max-tokens 16384 --reflection-timeout 3600)
|
|
30
|
+
[[ -n $PROPOSER_URL ]] && proposer+=(--reflection-base-url $PROPOSER_URL)
|
|
31
|
+
[[ -n $PROPOSER_OPTIONS ]] && proposer+=(--reflection-options $PROPOSER_OPTIONS)
|
|
32
|
+
judge=(--backend ruling --max-workers 1 --base-url $JUDGE --engine $STOCKFISH)
|
|
33
|
+
|
|
34
|
+
deadline=$(( $(date +%s) + HOURS * 3600 ))
|
|
35
|
+
note() { print -r -- "$(date '+%F %T') $*" | tee -a $LOGS/status.txt; }
|
|
36
|
+
|
|
37
|
+
caffeinate -dimsu -w $$ &
|
|
38
|
+
print $$ >$LOGS/driver.pid
|
|
39
|
+
runs=()
|
|
40
|
+
halt() {
|
|
41
|
+
trap - TERM INT
|
|
42
|
+
# Each search stops after its current round and keeps its checkpoint; running this script
|
|
43
|
+
# again resumes them. A second signal ends the searches at once.
|
|
44
|
+
note "stopping after the current round"
|
|
45
|
+
for search in $LOGS/*/search; do [[ -d $search ]] && touch $search/gepa.stop; done
|
|
46
|
+
trap 'pkill -f "^[^ ]*python3 [^ ]*/rulesmith (optimize|evaluate) examples/chess/"' TERM INT
|
|
47
|
+
[[ -n ${stopper:-} ]] && kill $stopper 2>/dev/null
|
|
48
|
+
wait $runs
|
|
49
|
+
rm -f $LOGS/driver.pid
|
|
50
|
+
exit 143
|
|
51
|
+
}
|
|
52
|
+
trap halt TERM INT
|
|
53
|
+
|
|
54
|
+
search() {
|
|
55
|
+
local seed=$1 run=$LOGS/chess-$1
|
|
56
|
+
if [[ -f $run/report.json ]]; then
|
|
57
|
+
note "seed $seed: already done"
|
|
58
|
+
else
|
|
59
|
+
# A search that was interrupted resumes from its checkpoint; one that left none is set aside.
|
|
60
|
+
resume=()
|
|
61
|
+
if [[ -f $run/search/gepa_state.bin ]]; then resume=(--resume)
|
|
62
|
+
elif [[ -d $run ]]; then mv $run $run-failed-$(date +%s)
|
|
63
|
+
fi
|
|
64
|
+
note "seed $seed: searching, $BUDGET games"
|
|
65
|
+
uv run rulesmith optimize $TASK $START $judge --max-metric-calls $BUDGET --seed $seed \
|
|
66
|
+
$proposer $resume --output $run >>$LOGS/chess-$seed.log 2>&1 \
|
|
67
|
+
|| { note "seed $seed: search failed, see $LOGS/chess-$seed.log"; return; }
|
|
68
|
+
fi
|
|
69
|
+
if [[ ! -f $run/test.json ]]; then
|
|
70
|
+
note "seed $seed: scoring the winner on test"
|
|
71
|
+
uv run rulesmith evaluate $TASK $run/plan.json $judge --split test --output $run/test.json \
|
|
72
|
+
>>$LOGS/chess-$seed.log 2>&1 || note "seed $seed: test failed, see $LOGS/chess-$seed.log"
|
|
73
|
+
fi
|
|
74
|
+
note "seed $seed: done"
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
# Games take a core each and share nothing, so the seeds run at once.
|
|
78
|
+
for seed in $SEEDS; do search $seed & runs+=($!); done
|
|
79
|
+
( sleep $(( deadline - $(date +%s) )); note "deadline reached"; kill -TERM $$ ) &
|
|
80
|
+
stopper=$!
|
|
81
|
+
wait $runs
|
|
82
|
+
kill $stopper 2>/dev/null
|
|
83
|
+
rm -f $LOGS/driver.pid
|
|
84
|
+
uv run python3 benchmarks/fitness.py runs/doom runs/chess benchmarks/results/games.json \
|
|
85
|
+
docs/assets/chess-stockfish.svg --tasks chess \
|
|
86
|
+
&& note "all finished; chart rebuilt" || note "chart not rebuilt; see above"
|