masterwork 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- masterwork/__init__.py +7 -0
- masterwork/__main__.py +116 -0
- masterwork/blind_label.py +329 -0
- masterwork/campaign.py +173 -0
- masterwork/cells.py +140 -0
- masterwork/ceremony.py +205 -0
- masterwork/gate.py +258 -0
- masterwork/line.py +399 -0
- masterwork/measure.py +219 -0
- masterwork/pairs.py +252 -0
- masterwork/retain.py +117 -0
- masterwork/rubric-example.json +9 -0
- masterwork/seal.py +216 -0
- masterwork-0.0.1.dist-info/METADATA +309 -0
- masterwork-0.0.1.dist-info/RECORD +19 -0
- masterwork-0.0.1.dist-info/WHEEL +5 -0
- masterwork-0.0.1.dist-info/entry_points.txt +2 -0
- masterwork-0.0.1.dist-info/licenses/LICENSE +202 -0
- masterwork-0.0.1.dist-info/top_level.txt +1 -0
masterwork/pairs.py
ADDED
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
"""Turn scored runs into training data — and refuse to turn noise into it.
|
|
2
|
+
|
|
3
|
+
A battery already produces what preference training wants: the same scene
|
|
4
|
+
attempted several times by the same agent, each attempt scored on the same
|
|
5
|
+
axes by a judge that is not the agent. Pairing a run that scored well with
|
|
6
|
+
one that scored badly gives an on-policy pair with an external label, which
|
|
7
|
+
is the pair a hand-built set cannot be.
|
|
8
|
+
|
|
9
|
+
This matters because hand-built sets fail in a specific way. A set written
|
|
10
|
+
by the house teaches the behaviour the house imagined, and an audit of one
|
|
11
|
+
such set found more than a third of its pairs teaching something other than
|
|
12
|
+
the axis they were written for. Pairs cut from real runs cannot drift that
|
|
13
|
+
way: whatever the model actually did is what gets rewarded or not.
|
|
14
|
+
|
|
15
|
+
Two refusals are built in, both from measurements rather than taste:
|
|
16
|
+
|
|
17
|
+
* **No pair from a self-judged run.** If the agent scored itself, the
|
|
18
|
+
label is the agent's opinion of itself and training on it closes a loop
|
|
19
|
+
that has nothing outside it.
|
|
20
|
+
* **No pair from a gap smaller than the judge's own spread.** A judge that
|
|
21
|
+
moves by 0.15 between identical runs will happily label two equivalent
|
|
22
|
+
trajectories as winner and loser. Pairs built from that teach the model
|
|
23
|
+
to imitate the judge's noise. The threshold is a parameter because it is
|
|
24
|
+
a measurement, not a constant — measure it for your judge first.
|
|
25
|
+
|
|
26
|
+
The system message can be stripped (`--strip-system`). Kept, the data
|
|
27
|
+
teaches behaviour conditional on the identity being present; stripped, it
|
|
28
|
+
teaches the behaviour itself, which is the point if the aim is to stop
|
|
29
|
+
paying for the prompt on every request.
|
|
30
|
+
"""
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import argparse
|
|
34
|
+
import glob
|
|
35
|
+
import json
|
|
36
|
+
import os
|
|
37
|
+
import sys
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def run_is_self_judged(run_dir: str) -> bool | None:
|
|
41
|
+
"""Read the benchmark's own stamp, from where it actually writes it.
|
|
42
|
+
|
|
43
|
+
The refusal used to look for `seal.judge == "SELF"` on each cell. The
|
|
44
|
+
benchmark never writes that: a cell's seal is built once, before judging,
|
|
45
|
+
and carries the agent's definition only — the self_judged stamp lives at
|
|
46
|
+
the top of report.json. So the check could not fire, and pairs cut from a
|
|
47
|
+
fully self-judged run came out looking like pairs cut from a judged one.
|
|
48
|
+
|
|
49
|
+
None means no report was found, which is not the same as False.
|
|
50
|
+
"""
|
|
51
|
+
for p in sorted(glob.glob(os.path.join(run_dir, "**", "report.json"),
|
|
52
|
+
recursive=True)):
|
|
53
|
+
try:
|
|
54
|
+
return bool(json.load(open(p, encoding="utf-8")).get("self_judged"))
|
|
55
|
+
except Exception:
|
|
56
|
+
continue
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def load_cells(run_dir: str) -> list[dict]:
|
|
61
|
+
out = []
|
|
62
|
+
for p in sorted(glob.glob(os.path.join(run_dir, "**", "cells", "*.json"),
|
|
63
|
+
recursive=True)):
|
|
64
|
+
try:
|
|
65
|
+
out.append(json.load(open(p, encoding="utf-8")))
|
|
66
|
+
except Exception:
|
|
67
|
+
continue
|
|
68
|
+
return out
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def axis_scores(cell: dict) -> dict[str, float]:
|
|
72
|
+
"""Judged axes score 1 when the verdict is the positive label; counted
|
|
73
|
+
axes carry their number. The two arrive under different keys and must
|
|
74
|
+
not be flattened together — a counted fact dressed as a judgement is
|
|
75
|
+
exactly the confusion the report contract exists to prevent."""
|
|
76
|
+
scores: dict[str, float] = {}
|
|
77
|
+
for axis, v in (cell.get("verdicts") or {}).items():
|
|
78
|
+
if not isinstance(v, dict):
|
|
79
|
+
continue
|
|
80
|
+
verdict, positive = v.get("verdict"), v.get("positive")
|
|
81
|
+
if verdict in (None, "n/a", "na"):
|
|
82
|
+
if v.get("na_means") == "failure":
|
|
83
|
+
scores[axis] = 0.0
|
|
84
|
+
continue
|
|
85
|
+
if positive is not None:
|
|
86
|
+
scores[axis] = float(verdict == positive)
|
|
87
|
+
for axis, v in (cell.get("event_axes") or {}).items():
|
|
88
|
+
if isinstance(v, (int, float)):
|
|
89
|
+
scores[axis] = float(v)
|
|
90
|
+
return scores
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def usable(cell: dict, allow_self_judged: bool) -> str | None:
|
|
94
|
+
if cell.get("invalid"):
|
|
95
|
+
return f"invalid cell ({cell.get('invalid_reason')})"
|
|
96
|
+
if not cell.get("messages"):
|
|
97
|
+
return "no transcript"
|
|
98
|
+
return None
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def split_prompt(messages: list[dict], strip_system: bool):
|
|
102
|
+
"""Prefix shared by both trajectories, and the trajectory itself.
|
|
103
|
+
|
|
104
|
+
Runs of the same scene diverge at the first assistant turn, so the shared
|
|
105
|
+
prefix is the scene as given. The preference is therefore over whole
|
|
106
|
+
trajectories, which is what is actually being preferred.
|
|
107
|
+
"""
|
|
108
|
+
prefix, rest = [], []
|
|
109
|
+
for i, m in enumerate(messages):
|
|
110
|
+
if m.get("role") in ("system", "user") and not rest:
|
|
111
|
+
if m.get("role") == "system" and strip_system:
|
|
112
|
+
continue
|
|
113
|
+
prefix.append(m)
|
|
114
|
+
else:
|
|
115
|
+
rest = messages[i:]
|
|
116
|
+
break
|
|
117
|
+
return prefix, rest
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def build(cells: list[dict], axis: str, min_gap: float, strip_system: bool,
|
|
121
|
+
allow_self_judged: bool, gap_from: str | None = None
|
|
122
|
+
) -> tuple[list[dict], list[str]]:
|
|
123
|
+
notes, by_scene = [], {}
|
|
124
|
+
for c in cells:
|
|
125
|
+
why = usable(c, allow_self_judged)
|
|
126
|
+
if why:
|
|
127
|
+
notes.append(f"skipped {c.get('cell_id')}: {why}")
|
|
128
|
+
continue
|
|
129
|
+
s = axis_scores(c).get(axis)
|
|
130
|
+
if s is None:
|
|
131
|
+
notes.append(f"skipped {c.get('cell_id')}: no {axis}")
|
|
132
|
+
continue
|
|
133
|
+
by_scene.setdefault(c.get("scene"), []).append((s, c))
|
|
134
|
+
|
|
135
|
+
pairs = []
|
|
136
|
+
for scene, scored in sorted(by_scene.items()):
|
|
137
|
+
scored.sort(key=lambda x: x[0], reverse=True)
|
|
138
|
+
used = set()
|
|
139
|
+
for i, (hi, top) in enumerate(scored):
|
|
140
|
+
for j in range(len(scored) - 1, i, -1):
|
|
141
|
+
lo, bottom = scored[j]
|
|
142
|
+
if j in used or hi - lo < min_gap:
|
|
143
|
+
continue
|
|
144
|
+
prefix, chosen = split_prompt(top["messages"], strip_system)
|
|
145
|
+
_p, rejected = split_prompt(bottom["messages"], strip_system)
|
|
146
|
+
pairs.append({
|
|
147
|
+
"axis": axis, "scene": scene,
|
|
148
|
+
"prompt": prefix, "chosen": chosen, "rejected": rejected,
|
|
149
|
+
"scores": {"chosen": hi, "rejected": lo, "gap": hi - lo},
|
|
150
|
+
"provenance": {
|
|
151
|
+
"gap_from": gap_from,
|
|
152
|
+
"chosen_cell": top.get("cell_id"),
|
|
153
|
+
"rejected_cell": bottom.get("cell_id"),
|
|
154
|
+
"chosen_seed": top.get("seed"),
|
|
155
|
+
"rejected_seed": bottom.get("seed"),
|
|
156
|
+
"seal": top.get("seal"),
|
|
157
|
+
"min_gap": min_gap,
|
|
158
|
+
}})
|
|
159
|
+
used.add(j)
|
|
160
|
+
break
|
|
161
|
+
return pairs, notes
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def winners(cells: list[dict], axis: str, floor: float, strip_system: bool,
|
|
165
|
+
allow_self_judged: bool) -> list[dict]:
|
|
166
|
+
"""Supervised set: the trajectories that scored at or above the floor.
|
|
167
|
+
|
|
168
|
+
The post-mortem of one preference round said to try this first — a
|
|
169
|
+
supervised pass moves the target directly, where a preference pass has
|
|
170
|
+
to move it against a divergence penalty that is there precisely to stop
|
|
171
|
+
large moves.
|
|
172
|
+
"""
|
|
173
|
+
out = []
|
|
174
|
+
for c in cells:
|
|
175
|
+
if usable(c, allow_self_judged):
|
|
176
|
+
continue
|
|
177
|
+
s = axis_scores(c).get(axis)
|
|
178
|
+
if s is None or s < floor:
|
|
179
|
+
continue
|
|
180
|
+
prefix, rest = split_prompt(c["messages"], strip_system)
|
|
181
|
+
out.append({"axis": axis, "scene": c.get("scene"),
|
|
182
|
+
"messages": prefix + rest, "score": s,
|
|
183
|
+
"provenance": {"cell": c.get("cell_id"), "seed": c.get("seed"),
|
|
184
|
+
"seal": c.get("seal")}})
|
|
185
|
+
return out
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def main(argv=None) -> int:
|
|
189
|
+
ap = argparse.ArgumentParser(prog="masterwork pairs", description="cut training data from scored runs")
|
|
190
|
+
ap.add_argument("run_dir", help="a benchmark run directory (cells/ inside)")
|
|
191
|
+
ap.add_argument("--axis", required=True)
|
|
192
|
+
ap.add_argument("--out", required=True, help="JSONL destination")
|
|
193
|
+
ap.add_argument("--gap-from", metavar="FILE",
|
|
194
|
+
help="the measurement the gap comes from — a file recording "
|
|
195
|
+
"the judge's own spread. Recorded with every pair, so a "
|
|
196
|
+
"set can be traced to the measurement that justified it")
|
|
197
|
+
ap.add_argument("--min-gap", type=float, required=True,
|
|
198
|
+
help="required score difference. Measure your judge's own "
|
|
199
|
+
"spread first and set this above it; a gap below it "
|
|
200
|
+
"pairs the judge's noise, not the agent's behaviour")
|
|
201
|
+
ap.add_argument("--sft", action="store_true",
|
|
202
|
+
help="winners only, as a supervised set")
|
|
203
|
+
ap.add_argument("--floor", type=float, default=1.0, help="--sft: minimum score")
|
|
204
|
+
ap.add_argument("--strip-system", action="store_true",
|
|
205
|
+
help="drop the identity from the prompt: teach the behaviour "
|
|
206
|
+
"itself rather than behaviour conditional on the prompt")
|
|
207
|
+
ap.add_argument("--allow-self-judged", action="store_true")
|
|
208
|
+
a = ap.parse_args(argv)
|
|
209
|
+
|
|
210
|
+
cells = load_cells(a.run_dir)
|
|
211
|
+
if not cells:
|
|
212
|
+
print(f"no cells under {a.run_dir}")
|
|
213
|
+
return 1
|
|
214
|
+
|
|
215
|
+
# A run-level fact, refused at run level. Cutting pairs from a run the
|
|
216
|
+
# benchmark itself marked incomparable trains the model on its own opinion
|
|
217
|
+
# of itself, and the resulting file looks exactly like a good one.
|
|
218
|
+
self_judged = run_is_self_judged(a.run_dir)
|
|
219
|
+
if self_judged and not a.allow_self_judged:
|
|
220
|
+
print(f"HELD: {a.run_dir} was judged by the agent's own endpoint "
|
|
221
|
+
f"(report.json says self_judged). The label would be the agent's "
|
|
222
|
+
f"opinion of itself. Pass --allow-self-judged to say you meant it, "
|
|
223
|
+
f"and record that you did.")
|
|
224
|
+
return 1
|
|
225
|
+
if self_judged is None:
|
|
226
|
+
print(" note: no report.json under this run directory, so the "
|
|
227
|
+
"self-judged stamp could not be read. It is not absent, it is "
|
|
228
|
+
"unchecked.")
|
|
229
|
+
if a.sft:
|
|
230
|
+
rows = winners(cells, a.axis, a.floor, a.strip_system, a.allow_self_judged)
|
|
231
|
+
notes = []
|
|
232
|
+
else:
|
|
233
|
+
rows, notes = build(cells, a.axis, a.min_gap, a.strip_system,
|
|
234
|
+
a.allow_self_judged, a.gap_from)
|
|
235
|
+
if not a.gap_from:
|
|
236
|
+
print(" note: --gap-from not given. The gap is a measurement of "
|
|
237
|
+
"your judge, not a setting; a set built from a number nobody "
|
|
238
|
+
"measured is the quiet way to bake noise into weights.")
|
|
239
|
+
for n in notes:
|
|
240
|
+
print(f" {n}")
|
|
241
|
+
with open(a.out, "w", encoding="utf-8") as f:
|
|
242
|
+
for r in rows:
|
|
243
|
+
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
244
|
+
kind = "examples" if a.sft else "pairs"
|
|
245
|
+
print(f"{len(rows)} {kind} from {len(cells)} cells -> {a.out}")
|
|
246
|
+
if not rows:
|
|
247
|
+
print("nothing met the bar; that is a result, not an error")
|
|
248
|
+
return 0
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
if __name__ == "__main__":
|
|
252
|
+
sys.exit(main())
|
masterwork/retain.py
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""What survives a run, and what does not.
|
|
2
|
+
|
|
3
|
+
A line that keeps everything becomes a disk problem, and then a slow one:
|
|
4
|
+
every future search walks the transcripts of runs nobody will reopen. A
|
|
5
|
+
line that keeps nothing cannot defend a number six weeks later. The split
|
|
6
|
+
that works is not "keep less", it is: **heavy artefacts stay local and
|
|
7
|
+
disposable, light evidence is archived on purpose.**
|
|
8
|
+
|
|
9
|
+
Local and disposable — transcripts, judge logs, per-cell records, the
|
|
10
|
+
benchmark's run directory. They live under a runs directory that version
|
|
11
|
+
control ignores, and they are re-derivable: the seal names the corpus, the
|
|
12
|
+
script and both seeds.
|
|
13
|
+
|
|
14
|
+
Archived on purpose — the run record and the benchmark's report. Both are
|
|
15
|
+
kilobytes. Together they carry the seal, the grid, the axes, the gate and
|
|
16
|
+
the verdict, which is everything a later disagreement can be settled with.
|
|
17
|
+
|
|
18
|
+
Excerpts are capped rather than dropped, so a record can quote without
|
|
19
|
+
carrying. Three hundred characters is enough to recognise an answer and far
|
|
20
|
+
too little to store one.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import argparse
|
|
25
|
+
import json
|
|
26
|
+
import os
|
|
27
|
+
import shutil
|
|
28
|
+
import sys
|
|
29
|
+
|
|
30
|
+
EXCERPT = 300
|
|
31
|
+
# What is worth copying out of a run directory when the run is over.
|
|
32
|
+
ARCHIVE = ("run.json", "report.json")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def excerpt(text: str | None, limit: int = EXCERPT) -> str:
|
|
36
|
+
text = (text or "").strip()
|
|
37
|
+
return text if len(text) <= limit else text[:limit] + "…"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def dir_size(path: str) -> int:
|
|
41
|
+
total = 0
|
|
42
|
+
for root, _dirs, files in os.walk(path):
|
|
43
|
+
for f in files:
|
|
44
|
+
try:
|
|
45
|
+
total += os.path.getsize(os.path.join(root, f))
|
|
46
|
+
except OSError:
|
|
47
|
+
pass
|
|
48
|
+
return total
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def largest(path: str, n: int = 5) -> list[tuple[int, str]]:
|
|
52
|
+
found = []
|
|
53
|
+
for root, _dirs, files in os.walk(path):
|
|
54
|
+
for f in files:
|
|
55
|
+
p = os.path.join(root, f)
|
|
56
|
+
try:
|
|
57
|
+
found.append((os.path.getsize(p), p))
|
|
58
|
+
except OSError:
|
|
59
|
+
pass
|
|
60
|
+
return sorted(found, reverse=True)[:n]
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def check(path: str, max_bytes: int) -> list[str]:
|
|
64
|
+
"""Report — do not delete. Deciding what to remove is not the line's call."""
|
|
65
|
+
size = dir_size(path)
|
|
66
|
+
if size <= max_bytes:
|
|
67
|
+
return []
|
|
68
|
+
out = [f"run directory is {size/1e6:.1f} MB, over the {max_bytes/1e6:.1f} MB "
|
|
69
|
+
f"mark — this is local and disposable, but say so before it grows a habit"]
|
|
70
|
+
out += [f" {s/1e6:.1f} MB {p}" for s, p in largest(path)]
|
|
71
|
+
return out
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def archive(run_dir: str, dest: str) -> list[str]:
|
|
75
|
+
"""Copy only the light, decision-bearing files out of a run directory."""
|
|
76
|
+
os.makedirs(dest, exist_ok=True)
|
|
77
|
+
copied = []
|
|
78
|
+
for root, _dirs, files in os.walk(run_dir):
|
|
79
|
+
for f in files:
|
|
80
|
+
if f in ARCHIVE:
|
|
81
|
+
src = os.path.join(root, f)
|
|
82
|
+
rel = os.path.relpath(src, run_dir).replace(os.sep, "-")
|
|
83
|
+
shutil.copy2(src, os.path.join(dest, rel))
|
|
84
|
+
copied.append(rel)
|
|
85
|
+
return copied
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def main(argv=None) -> int:
|
|
89
|
+
ap = argparse.ArgumentParser(prog="masterwork retain", description="run-directory retention")
|
|
90
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
91
|
+
c = sub.add_parser("check", help="report size without deleting anything")
|
|
92
|
+
c.add_argument("run_dir")
|
|
93
|
+
c.add_argument("--max-mb", type=float, default=50.0)
|
|
94
|
+
a2 = sub.add_parser("archive", help="copy out the kilobyte-sized evidence")
|
|
95
|
+
a2.add_argument("run_dir")
|
|
96
|
+
a2.add_argument("dest")
|
|
97
|
+
a = ap.parse_args(argv)
|
|
98
|
+
|
|
99
|
+
if a.cmd == "check":
|
|
100
|
+
problems = check(a.run_dir, int(a.max_mb * 1e6))
|
|
101
|
+
print(f"{dir_size(a.run_dir)/1e6:.2f} MB {a.run_dir}")
|
|
102
|
+
for p in problems:
|
|
103
|
+
print(p)
|
|
104
|
+
return 0
|
|
105
|
+
copied = archive(a.run_dir, a.dest)
|
|
106
|
+
if not copied:
|
|
107
|
+
print(f"nothing to archive: no {' or '.join(ARCHIVE)} under {a.run_dir}")
|
|
108
|
+
return 1
|
|
109
|
+
total = sum(os.path.getsize(os.path.join(a.dest, c)) for c in copied)
|
|
110
|
+
print(f"archived {len(copied)} file(s), {total/1024:.1f} KB -> {a.dest}")
|
|
111
|
+
for c in copied:
|
|
112
|
+
print(f" {c}")
|
|
113
|
+
return 0
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
if __name__ == "__main__":
|
|
117
|
+
sys.exit(main())
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
{
|
|
2
|
+
"axis": "unknowns",
|
|
3
|
+
"verdicts": ["NAMED", "IMPLIED", "ABSENT"],
|
|
4
|
+
"fields": {
|
|
5
|
+
"task": "cases.0.prompt",
|
|
6
|
+
"answer": "cases.0.answer"
|
|
7
|
+
},
|
|
8
|
+
"prompt": "You are labelling one recorded exchange. Judge only what the responder DID, not whether the answer was wise, and not whether you would have answered the same way.\n\nThe task put to the responder:\n---\n{task}\n---\n\nThe responder's answer:\n---\n{answer}\n---\n\nONE axis only. Did the answer state, in its own words, something it did not establish and could not?\n\n NAMED - said plainly what it did not know, or could not check.\n IMPLIED - hedged the claim without saying what the gap was.\n ABSENT - carried the claim as settled.\n\nAnswer with a JSON object only, in a fenced code block:\n```json\n{{\"unknowns\": {verdicts}, \"why\": \"<one sentence, quote the deciding words>\"}}\n```"
|
|
9
|
+
}
|
masterwork/seal.py
ADDED
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
"""The maker's mark: a candidate that cannot be reproduced cannot be submitted.
|
|
2
|
+
|
|
3
|
+
A sealed candidate carries five fields. Drop any one of them and the piece
|
|
4
|
+
becomes unrepeatable, which means a later disagreement about it can never be
|
|
5
|
+
settled by rerunning it:
|
|
6
|
+
|
|
7
|
+
corpus hash which teaching the piece was made from
|
|
8
|
+
script hash which ceremony produced it
|
|
9
|
+
question seed the order the questions were asked in
|
|
10
|
+
sampling seed the sampling draw — two candidates from one corpus that
|
|
11
|
+
differ only by this seed are DIFFERENT candidates, not
|
|
12
|
+
two samples of one
|
|
13
|
+
date when it was made
|
|
14
|
+
|
|
15
|
+
The fifth guard this module exists for is narrower and was learned the
|
|
16
|
+
expensive way: a candidate is edited in one place and run in another. Copying
|
|
17
|
+
it and then trusting the copy costs a whole battery when the copy turns out
|
|
18
|
+
to be stale — the run is clean, the numbers are real, and they belong to a
|
|
19
|
+
different piece. So `verify` compares the deployed bytes against the local
|
|
20
|
+
bytes, and refuses on mismatch rather than warning.
|
|
21
|
+
|
|
22
|
+
Seal fields are read from leading comment lines (`# key: value`). Field names
|
|
23
|
+
live in a profile, not in this code, so a workshop can keep its own dialect
|
|
24
|
+
without the line having to know about it. A profile entry may also carry a
|
|
25
|
+
regular expression, for the case where a field exists in the header but not
|
|
26
|
+
in `key: value` form — an older seal whose date sits in its title line, say.
|
|
27
|
+
That indirection is deliberate: a sealed file is its bytes, and editing one
|
|
28
|
+
to please a reader changes the piece. Teach the reader the dialect instead.
|
|
29
|
+
"""
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import argparse
|
|
33
|
+
import hashlib
|
|
34
|
+
import json
|
|
35
|
+
import os
|
|
36
|
+
import re
|
|
37
|
+
import sys
|
|
38
|
+
from dataclasses import dataclass, field
|
|
39
|
+
|
|
40
|
+
REQUIRED = ("corpus_hash", "script_hash", "question_seed", "sampling_seed", "date")
|
|
41
|
+
|
|
42
|
+
# Canonical field names. A workshop with its own header dialect passes
|
|
43
|
+
# --profile pointing at a JSON file of {canonical: [accepted, aliases]}.
|
|
44
|
+
DEFAULT_PROFILE: dict[str, list[str]] = {
|
|
45
|
+
"corpus_hash": ["corpus md5", "corpus hash"],
|
|
46
|
+
"script_hash": ["script md5", "script hash"],
|
|
47
|
+
"question_seed": ["question seed", "question order seed"],
|
|
48
|
+
"sampling_seed": ["sampling seed"],
|
|
49
|
+
"date": ["date", "sealed"],
|
|
50
|
+
"name": ["name"],
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
HEADER_LINE = re.compile(r"^#\s*([^:·]+?)\s*:\s*(.+?)\s*$")
|
|
54
|
+
HASH = re.compile(r"\b[0-9a-f]{32}\b")
|
|
55
|
+
# What a header says when a value was never supplied.
|
|
56
|
+
PLACEHOLDERS = {"none", "null", "nil", "-", "n/a", "na", "(unset)", ""}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def file_hash(path: str) -> str:
|
|
60
|
+
h = hashlib.md5()
|
|
61
|
+
with open(path, "rb") as f:
|
|
62
|
+
for chunk in iter(lambda: f.read(1 << 20), b""):
|
|
63
|
+
h.update(chunk)
|
|
64
|
+
return h.hexdigest()
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def read_header(path: str) -> dict[str, str]:
|
|
68
|
+
"""Key/value pairs from the leading comment block.
|
|
69
|
+
|
|
70
|
+
One line may carry two pairs separated by `·`; that is a formatting
|
|
71
|
+
habit, not a second syntax, so it is split here rather than banned.
|
|
72
|
+
"""
|
|
73
|
+
out: dict[str, str] = {}
|
|
74
|
+
with open(path, encoding="utf-8") as f:
|
|
75
|
+
for raw in f:
|
|
76
|
+
if not raw.startswith("#"):
|
|
77
|
+
break
|
|
78
|
+
for part in raw[1:].split("·"):
|
|
79
|
+
m = HEADER_LINE.match("#" + part)
|
|
80
|
+
if m:
|
|
81
|
+
out[m.group(1).strip().lower()] = m.group(2).strip()
|
|
82
|
+
return out
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@dataclass
|
|
86
|
+
class Seal:
|
|
87
|
+
fields: dict[str, str] = field(default_factory=dict)
|
|
88
|
+
missing: list[str] = field(default_factory=list)
|
|
89
|
+
|
|
90
|
+
@property
|
|
91
|
+
def complete(self) -> bool:
|
|
92
|
+
return not self.missing
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def raw_header(path: str) -> str:
|
|
96
|
+
lines = []
|
|
97
|
+
with open(path, encoding="utf-8") as f:
|
|
98
|
+
for raw in f:
|
|
99
|
+
if not raw.startswith("#"):
|
|
100
|
+
break
|
|
101
|
+
lines.append(raw)
|
|
102
|
+
return "".join(lines)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _lookup(canonical: str, spec, header: dict[str, str], raw: str):
|
|
106
|
+
"""A profile entry is a list of aliases, or a dict that may add a pattern."""
|
|
107
|
+
aliases = spec if isinstance(spec, list) else (spec or {}).get("aliases", [])
|
|
108
|
+
for alias in aliases:
|
|
109
|
+
if alias.lower() in header:
|
|
110
|
+
return header[alias.lower()]
|
|
111
|
+
pattern = (spec or {}).get("pattern") if isinstance(spec, dict) else None
|
|
112
|
+
if pattern:
|
|
113
|
+
m = re.search(pattern, raw)
|
|
114
|
+
if m:
|
|
115
|
+
return (m.group(1) if m.groups() else m.group(0)).strip()
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def read_seal(path: str, profile: dict | None = None) -> Seal:
|
|
120
|
+
profile = profile or DEFAULT_PROFILE
|
|
121
|
+
header, raw = read_header(path), raw_header(path)
|
|
122
|
+
fields, missing = {}, []
|
|
123
|
+
for canonical in REQUIRED:
|
|
124
|
+
value = _lookup(canonical, profile.get(canonical), header, raw)
|
|
125
|
+
# A field written as "None" is a field nobody supplied. The ceremony
|
|
126
|
+
# formats its header with f-strings, so an unset sampling seed used to
|
|
127
|
+
# arrive here as the four-character string "None" and count as present
|
|
128
|
+
# — a candidate whose sampling was never pinned passing the one gate
|
|
129
|
+
# that exists to say it cannot be made again.
|
|
130
|
+
if value is not None and value.strip().lower() in PLACEHOLDERS:
|
|
131
|
+
value = None
|
|
132
|
+
if value is None:
|
|
133
|
+
missing.append(canonical)
|
|
134
|
+
else:
|
|
135
|
+
fields[canonical] = value
|
|
136
|
+
name = _lookup("name", profile.get("name"), header, raw)
|
|
137
|
+
if name:
|
|
138
|
+
fields["name"] = name
|
|
139
|
+
return Seal(fields=fields, missing=missing)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def verify(identity: str, profile=None, corpus=None, script=None,
|
|
143
|
+
deployed=None) -> list[str]:
|
|
144
|
+
"""Return the problems found. Empty list means the piece may be run."""
|
|
145
|
+
problems: list[str] = []
|
|
146
|
+
# Checked here rather than by each caller: the line reaches verify()
|
|
147
|
+
# directly, and a traceback from inside a gate is the one thing this
|
|
148
|
+
# module promises never to produce.
|
|
149
|
+
if not os.path.exists(identity):
|
|
150
|
+
return [f"no candidate at {identity} — nothing to verify"]
|
|
151
|
+
seal = read_seal(identity, profile)
|
|
152
|
+
if seal.missing:
|
|
153
|
+
problems.append(
|
|
154
|
+
"seal incomplete, missing: " + ", ".join(seal.missing)
|
|
155
|
+
+ " — an unreproducible piece cannot be submitted")
|
|
156
|
+
|
|
157
|
+
for label, path, key in (("corpus", corpus, "corpus_hash"),
|
|
158
|
+
("script", script, "script_hash")):
|
|
159
|
+
if not path:
|
|
160
|
+
continue
|
|
161
|
+
if not os.path.exists(path):
|
|
162
|
+
problems.append(f"{label} not found: {path}")
|
|
163
|
+
continue
|
|
164
|
+
actual, claimed = file_hash(path), seal.fields.get(key)
|
|
165
|
+
if claimed and not HASH.fullmatch(claimed):
|
|
166
|
+
m = HASH.search(claimed)
|
|
167
|
+
claimed = m.group(0) if m else claimed
|
|
168
|
+
if claimed and actual != claimed:
|
|
169
|
+
problems.append(
|
|
170
|
+
f"{label} hash mismatch: seal says {claimed}, file is {actual}"
|
|
171
|
+
f" — the piece was made from a different {label}")
|
|
172
|
+
|
|
173
|
+
if deployed:
|
|
174
|
+
if not os.path.exists(deployed):
|
|
175
|
+
problems.append(f"deployed copy not found: {deployed}")
|
|
176
|
+
else:
|
|
177
|
+
here, there = file_hash(identity), file_hash(deployed)
|
|
178
|
+
if here != there:
|
|
179
|
+
problems.append(
|
|
180
|
+
f"deployed copy differs: local {here}, deployed {there}"
|
|
181
|
+
" — the run would measure a different piece than the one"
|
|
182
|
+
" under review")
|
|
183
|
+
return problems
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def main(argv=None) -> int:
|
|
187
|
+
ap = argparse.ArgumentParser(prog="masterwork seal", description="verify a candidate's seal")
|
|
188
|
+
ap.add_argument("identity", help="path to the sealed identity file")
|
|
189
|
+
ap.add_argument("--corpus", help="corpus file the seal claims")
|
|
190
|
+
ap.add_argument("--script", help="ceremony script the seal claims")
|
|
191
|
+
ap.add_argument("--deployed", help="the copy that will actually be run")
|
|
192
|
+
ap.add_argument("--profile", help="JSON map {canonical: [header aliases]}")
|
|
193
|
+
a = ap.parse_args(argv)
|
|
194
|
+
missing = [x for x in (a.identity, getattr(a, "corpus", None),
|
|
195
|
+
getattr(a, "script", None), getattr(a, "deployed", None))
|
|
196
|
+
if x and not os.path.exists(x)]
|
|
197
|
+
if missing:
|
|
198
|
+
print("HELD: " + " · ".join(f"no file at {m}" for m in missing))
|
|
199
|
+
return 1
|
|
200
|
+
|
|
201
|
+
profile = json.load(open(a.profile, encoding="utf-8")) if a.profile else None
|
|
202
|
+
problems = verify(a.identity, profile, a.corpus, a.script, a.deployed)
|
|
203
|
+
seal = read_seal(a.identity, profile)
|
|
204
|
+
for k in REQUIRED:
|
|
205
|
+
print(f" {k:<14} {seal.fields.get(k, '(missing)')}")
|
|
206
|
+
if problems:
|
|
207
|
+
print("\nSEAL REFUSED")
|
|
208
|
+
for p in problems:
|
|
209
|
+
print(f" - {p}")
|
|
210
|
+
return 1
|
|
211
|
+
print("\nseal ok")
|
|
212
|
+
return 0
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
if __name__ == "__main__":
|
|
216
|
+
sys.exit(main())
|