rulesmith 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rulesmith/__init__.py +1 -0
- rulesmith/ablate.py +99 -0
- rulesmith/arena.py +172 -0
- rulesmith/bench.py +1068 -0
- rulesmith/calibrate.py +457 -0
- rulesmith/chat_judge.py +205 -0
- rulesmith/chess.py +526 -0
- rulesmith/clef.py +66 -0
- rulesmith/cli.py +1234 -0
- rulesmith/diagram.py +226 -0
- rulesmith/doom.py +550 -0
- rulesmith/extract.py +77 -0
- rulesmith/grade.py +85 -0
- rulesmith/graph.py +975 -0
- rulesmith/label.py +67 -0
- rulesmith/level.py +389 -0
- rulesmith/maps.py +96 -0
- rulesmith/mine.py +313 -0
- rulesmith/optimize.py +931 -0
- rulesmith/rules.py +1017 -0
- rulesmith/runtime.py +711 -0
- rulesmith/serve.py +68 -0
- rulesmith/tuning.py +134 -0
- rulesmith-0.1.0.dist-info/METADATA +131 -0
- rulesmith-0.1.0.dist-info/RECORD +28 -0
- rulesmith-0.1.0.dist-info/WHEEL +4 -0
- rulesmith-0.1.0.dist-info/entry_points.txt +2 -0
- rulesmith-0.1.0.dist-info/licenses/LICENSE +21 -0
rulesmith/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Typed decision graphs, answered by a judge and searched with GEPA."""
|
rulesmith/ablate.py
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""What each input is worth: switch one off and count the decisions that change.
|
|
2
|
+
|
|
3
|
+
A reading that changes nothing is either a path the state never had, a rule an earlier one always
|
|
4
|
+
beats to the decision, or a threshold real data never trips. All three look identical from a score
|
|
5
|
+
-- the graph runs, decides something, and scores like a graph that simply decides badly.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
|
|
10
|
+
from rulesmith.graph import Call, Plan, Read, State, Sum
|
|
11
|
+
from rulesmith.rules import named_condition
|
|
12
|
+
from rulesmith.runtime import (
|
|
13
|
+
MISSING,
|
|
14
|
+
DecisionProgram,
|
|
15
|
+
ExecutionConfig,
|
|
16
|
+
Models,
|
|
17
|
+
Remembered,
|
|
18
|
+
read_path,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True)
|
|
23
|
+
class Effect:
|
|
24
|
+
"""One input's share of the decisions it changes when it is taken away."""
|
|
25
|
+
|
|
26
|
+
node: str
|
|
27
|
+
kind: str
|
|
28
|
+
changed: int
|
|
29
|
+
decided: int
|
|
30
|
+
reason: str = ""
|
|
31
|
+
|
|
32
|
+
@property
|
|
33
|
+
def share(self) -> float:
|
|
34
|
+
return self.changed / self.decided if self.decided else 0.0
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def decisions(program: DecisionProgram, states: list[State]) -> list[dict | None]:
|
|
38
|
+
"""Every output the graph gives each state, or None where it cannot decide at all. A graph
|
|
39
|
+
may decide more than a label, and a change to any output is a changed decision."""
|
|
40
|
+
decided = []
|
|
41
|
+
for state in states:
|
|
42
|
+
prediction = program(state=state)
|
|
43
|
+
decided.append(None if prediction.undecided else prediction.outputs)
|
|
44
|
+
return decided
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def effects(
|
|
48
|
+
plan: Plan,
|
|
49
|
+
states: list[State],
|
|
50
|
+
client=None,
|
|
51
|
+
model: str = "none",
|
|
52
|
+
execution: ExecutionConfig | None = None,
|
|
53
|
+
models: Models | None = None,
|
|
54
|
+
) -> list[Effect]:
|
|
55
|
+
"""Every read, judgment and name, ranked by how much the graph would change without it.
|
|
56
|
+
|
|
57
|
+
The judge is asked once per state for the baseline and never again: taking away a reading
|
|
58
|
+
cannot change what a judge said about the same state, so the replays reuse those answers and
|
|
59
|
+
a graph full of questions ablates for free."""
|
|
60
|
+
inputs = {}
|
|
61
|
+
for name, node in plan.nodes.items():
|
|
62
|
+
if isinstance(node, (Read, Call)):
|
|
63
|
+
inputs[name] = "read" if isinstance(node, Read) else "ask"
|
|
64
|
+
elif isinstance(node, Sum) or named_condition(node):
|
|
65
|
+
inputs[name] = "let"
|
|
66
|
+
judge = Remembered(client) if client is not None else None
|
|
67
|
+
models = (models or Models()).remembered()
|
|
68
|
+
execution = execution or ExecutionConfig()
|
|
69
|
+
baseline = decisions(DecisionProgram(plan, judge, model, execution, models=models), states)
|
|
70
|
+
found = []
|
|
71
|
+
for name, kind in inputs.items():
|
|
72
|
+
without = execution.model_copy(update={"suppress": (name,)})
|
|
73
|
+
suppressed = DecisionProgram(plan, judge, model, without, models=models)
|
|
74
|
+
changed = sum(
|
|
75
|
+
mine != theirs
|
|
76
|
+
for mine, theirs in zip(baseline, decisions(suppressed, states), strict=True)
|
|
77
|
+
)
|
|
78
|
+
found.append(
|
|
79
|
+
Effect(
|
|
80
|
+
node=name,
|
|
81
|
+
kind=kind,
|
|
82
|
+
changed=changed,
|
|
83
|
+
decided=len(states),
|
|
84
|
+
reason=blank(plan, name, states) if not changed else "",
|
|
85
|
+
)
|
|
86
|
+
)
|
|
87
|
+
return sorted(found, key=lambda effect: (-effect.changed, effect.node))
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def blank(plan: Plan, name: str, states: list[State]) -> str:
|
|
91
|
+
"""Why an input changed nothing, when the graph can tell. A read whose path is not in the
|
|
92
|
+
state is a different problem from a rule that never wins, and only one of them is a typo."""
|
|
93
|
+
node = plan.nodes[name]
|
|
94
|
+
if not isinstance(node, Read):
|
|
95
|
+
return "the graph decides the same without it"
|
|
96
|
+
present = sum(read_path(node, state) is not MISSING for state in states)
|
|
97
|
+
if not present:
|
|
98
|
+
return "this path is never in the state"
|
|
99
|
+
return f"present in {present} of {len(states)} states and changes no decision"
|
rulesmith/arena.py
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""Rank decision graphs by Elo from round-robin deathmatch duels."""
|
|
2
|
+
|
|
3
|
+
import math
|
|
4
|
+
import socket
|
|
5
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
6
|
+
from dataclasses import dataclass, replace
|
|
7
|
+
from itertools import combinations
|
|
8
|
+
from statistics import fmean
|
|
9
|
+
|
|
10
|
+
from rulesmith.doom import DEATHMATCH, DoomEnvironment, DoomTask, Episode, Episodes
|
|
11
|
+
from rulesmith.optimize import Outcome
|
|
12
|
+
from rulesmith.runtime import DecisionProgram
|
|
13
|
+
|
|
14
|
+
# Elo's conventions: 400 points is tenfold odds, and the field averages 1500.
|
|
15
|
+
ELO_SCALE = 400
|
|
16
|
+
ELO_MEAN = 1500
|
|
17
|
+
CONVERGED = 1e-10
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class Entrant:
|
|
22
|
+
name: str
|
|
23
|
+
program: DecisionProgram
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def free_port() -> int:
|
|
27
|
+
"""A UDP port nothing is using, for one game's host to listen on."""
|
|
28
|
+
with socket.socket(socket.AF_INET, socket.SOCK_DGRAM) as probe:
|
|
29
|
+
probe.bind(("127.0.0.1", 0))
|
|
30
|
+
return probe.getsockname()[1]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def play(
|
|
34
|
+
task: DoomTask,
|
|
35
|
+
program: DecisionProgram,
|
|
36
|
+
episode: int | Episode,
|
|
37
|
+
game_args: str = "",
|
|
38
|
+
frames: list | None = None,
|
|
39
|
+
) -> Outcome:
|
|
40
|
+
"""One seat's episode. A seat whose graph fails leaves the game, and its opponent plays on.
|
|
41
|
+
A frames list, when given, collects every tic this seat rendered."""
|
|
42
|
+
with DoomEnvironment(task, game_args=game_args, record=frames is not None) as env:
|
|
43
|
+
try:
|
|
44
|
+
return Episodes(task, env)(program, episode)
|
|
45
|
+
finally:
|
|
46
|
+
if frames is not None:
|
|
47
|
+
frames += env.frames
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def duel(
|
|
51
|
+
task: DoomTask,
|
|
52
|
+
a: DecisionProgram,
|
|
53
|
+
b: DecisionProgram,
|
|
54
|
+
episode: int | Episode,
|
|
55
|
+
host_first: bool,
|
|
56
|
+
frames: list | None = None,
|
|
57
|
+
) -> list[Outcome]:
|
|
58
|
+
"""Both seats' episodes of one deathmatch, a's first, each graph in its own networked game.
|
|
59
|
+
A frames list, when given, collects the tics a's seat rendered."""
|
|
60
|
+
port = free_port()
|
|
61
|
+
order = [a, b] if host_first else [b, a]
|
|
62
|
+
args = [f"-host 2 -port {port} {DEATHMATCH}", f"-join 127.0.0.1 -port {port}"]
|
|
63
|
+
seats = [frames if program is a else None for program in order]
|
|
64
|
+
# The host waits at init until its opponent joins, so both seats start together.
|
|
65
|
+
with ThreadPoolExecutor(2) as pool:
|
|
66
|
+
futures = [
|
|
67
|
+
pool.submit(play, task, program, episode, arg, seat)
|
|
68
|
+
for program, arg, seat in zip(order, args, seats, strict=True)
|
|
69
|
+
]
|
|
70
|
+
seats = [future.result() for future in futures]
|
|
71
|
+
return seats if host_first else seats[::-1]
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class Duels:
|
|
75
|
+
"""Scores a graph by its reward in a deathmatch against a fixed opponent per episode."""
|
|
76
|
+
|
|
77
|
+
def __init__(self, task: DoomTask, opponent: DecisionProgram):
|
|
78
|
+
self.task = task
|
|
79
|
+
self.opponent = opponent
|
|
80
|
+
|
|
81
|
+
def __call__(self, program: DecisionProgram, episode: int | Episode) -> Outcome:
|
|
82
|
+
# Alternating hosts by seed shares the joiner's latency handicap across episodes.
|
|
83
|
+
hosted = self.task.episode(episode).seed % 2 == 0
|
|
84
|
+
mine, theirs = duel(self.task, program, self.opponent, episode, hosted)
|
|
85
|
+
own, rival = mine.record["stats"]["frags"], theirs.record["stats"]["frags"]
|
|
86
|
+
result = "won" if own > rival else "lost" if own < rival else "drew"
|
|
87
|
+
feedback = f"{mine.trace['Feedback']} The opponent scored {rival} frags, so you {result}."
|
|
88
|
+
return replace(
|
|
89
|
+
mine,
|
|
90
|
+
record=mine.record | {"hosted": hosted, "opponent_frags": rival, "result": result},
|
|
91
|
+
trace=mine.trace | {"Feedback": feedback},
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def match(task: DoomTask, a: Entrant, b: Entrant, episode: int | Episode, host_first: bool) -> dict:
|
|
96
|
+
"""One deathmatch between two entrants, scored for the tournament."""
|
|
97
|
+
seats = duel(task, a.program, b.program, episode, host_first)
|
|
98
|
+
frags = [seat.record["stats"]["frags"] for seat in seats]
|
|
99
|
+
failed = [seat.error is not None for seat in seats]
|
|
100
|
+
if failed[0] != failed[1]:
|
|
101
|
+
score = [float(failed[1]), float(failed[0])]
|
|
102
|
+
elif frags[0] == frags[1]:
|
|
103
|
+
score = [0.5, 0.5]
|
|
104
|
+
else:
|
|
105
|
+
score = [float(frags[0] > frags[1]), float(frags[1] > frags[0])]
|
|
106
|
+
episode = task.episode(episode)
|
|
107
|
+
return {
|
|
108
|
+
"players": [a.name, b.name],
|
|
109
|
+
"seed": episode.seed,
|
|
110
|
+
"map": episode.map,
|
|
111
|
+
"host": (a if host_first else b).name,
|
|
112
|
+
"frags": frags,
|
|
113
|
+
"score": score,
|
|
114
|
+
"seats": [seat.record for seat in seats],
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def ratings(matches: list[dict], names: list[str]) -> dict[str, float]:
|
|
119
|
+
"""Bradley-Terry strengths fit to every result at once, so match order never matters,
|
|
120
|
+
reported on the Elo scale. Each pair also shares one virtual draw, which keeps an unbeaten
|
|
121
|
+
player's rating finite."""
|
|
122
|
+
points = {(x, y): 0.5 for x in names for y in names if x != y}
|
|
123
|
+
for record in matches:
|
|
124
|
+
(x, y), (earned_x, earned_y) = record["players"], record["score"]
|
|
125
|
+
points[x, y] += earned_x
|
|
126
|
+
points[y, x] += earned_y
|
|
127
|
+
strength = dict.fromkeys(names, 1.0)
|
|
128
|
+
while True:
|
|
129
|
+
# Hunter's minorization-maximization update, which converges to the maximum likelihood.
|
|
130
|
+
updated = {
|
|
131
|
+
x: sum(points[x, y] for y in names if y != x)
|
|
132
|
+
/ sum(
|
|
133
|
+
(points[x, y] + points[y, x]) / (strength[x] + strength[y]) for y in names if y != x
|
|
134
|
+
)
|
|
135
|
+
for x in names
|
|
136
|
+
}
|
|
137
|
+
center = math.exp(fmean(math.log(value) for value in updated.values()))
|
|
138
|
+
updated = {x: value / center for x, value in updated.items()}
|
|
139
|
+
done = max(abs(updated[x] - strength[x]) for x in names) < CONVERGED
|
|
140
|
+
strength = updated
|
|
141
|
+
if done:
|
|
142
|
+
break
|
|
143
|
+
return {x: ELO_MEAN + ELO_SCALE * math.log10(strength[x]) for x in names}
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def tournament(task: DoomTask, entrants: list[Entrant], episodes: list[int | Episode]) -> dict:
|
|
147
|
+
"""Every pair of entrants plays every episode, swapping hosts from one episode to the next."""
|
|
148
|
+
matches = []
|
|
149
|
+
for a, b in combinations(entrants, 2):
|
|
150
|
+
for index, episode in enumerate(episodes):
|
|
151
|
+
matches.append(match(task, a, b, episode, host_first=index % 2 == 0))
|
|
152
|
+
names = [entrant.name for entrant in entrants]
|
|
153
|
+
rated = ratings(matches, names)
|
|
154
|
+
records = {
|
|
155
|
+
name: {"wins": 0, "draws": 0, "losses": 0, "frags": 0, "frags_against": 0} for name in names
|
|
156
|
+
}
|
|
157
|
+
for record in matches:
|
|
158
|
+
for side, name in enumerate(record["players"]):
|
|
159
|
+
earned, own, rival = (
|
|
160
|
+
record["score"][side],
|
|
161
|
+
record["frags"][side],
|
|
162
|
+
record["frags"][1 - side],
|
|
163
|
+
)
|
|
164
|
+
records[name]["wins" if earned == 1 else "draws" if earned == 0.5 else "losses"] += 1
|
|
165
|
+
records[name]["frags"] += own
|
|
166
|
+
records[name]["frags_against"] += rival
|
|
167
|
+
standings = sorted(names, key=rated.get, reverse=True)
|
|
168
|
+
return {
|
|
169
|
+
"ratings": {name: round(rated[name], 1) for name in standings},
|
|
170
|
+
"records": {name: records[name] for name in standings},
|
|
171
|
+
"matches": matches,
|
|
172
|
+
}
|