ms408 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ms408/__init__.py +18 -0
- ms408/__main__.py +45 -0
- ms408/acquire.py +64 -0
- ms408/annotate/__init__.py +6 -0
- ms408/annotate/pipeline.py +223 -0
- ms408/annotate/qa.py +204 -0
- ms408/annotate/schema.py +217 -0
- ms408/benchmark.py +245 -0
- ms408/data/__init__.py +7 -0
- ms408/data/reference_bands.json +122 -0
- ms408/dataset.py +97 -0
- ms408/env.py +46 -0
- ms408/experiments/__init__.py +5 -0
- ms408/experiments/e10_third_rater.py +312 -0
- ms408/experiments/e11_style_control.py +184 -0
- ms408/experiments/e12_independent_rater.py +446 -0
- ms408/experiments/e13_function_content.py +290 -0
- ms408/experiments/e13b_function_content.py +181 -0
- ms408/experiments/e13c_function_content_corrected.py +156 -0
- ms408/experiments/e13d_band_robustness.py +169 -0
- ms408/experiments/e14_word_classes.py +210 -0
- ms408/experiments/e14b_word_class_robustness.py +166 -0
- ms408/experiments/e15_morphology.py +206 -0
- ms408/experiments/e15b_morphology_corrected.py +167 -0
- ms408/experiments/e17_ab_contrast.py +168 -0
- ms408/experiments/e18_corpus_completeness.py +107 -0
- ms408/experiments/e19_joint_signature.py +187 -0
- ms408/experiments/e19b_language_universality.py +139 -0
- ms408/experiments/e1_meaning_detector.py +253 -0
- ms408/experiments/e20_transposition_closure.py +148 -0
- ms408/experiments/e21_positional_generator.py +408 -0
- ms408/experiments/e22_generator_genericity.py +273 -0
- ms408/experiments/e23_reuse_generator.py +294 -0
- ms408/experiments/e24_typelevel_lexicon.py +287 -0
- ms408/experiments/e25_decoupled_ed1.py +359 -0
- ms408/experiments/e26_length_variance.py +277 -0
- ms408/experiments/e27_symbol_quantification.py +228 -0
- ms408/experiments/e28_angular_anchor.py +350 -0
- ms408/experiments/e29_naibbe_discriminators.py +265 -0
- ms408/experiments/e2_wordorder_confound.py +282 -0
- ms408/experiments/e30_cipher_reexamination.py +223 -0
- ms408/experiments/e31_harden_syntax.py +264 -0
- ms408/experiments/e32_reference_bands.py +158 -0
- ms408/experiments/e33_block_scale_di.py +281 -0
- ms408/experiments/e3_anchor_power.py +269 -0
- ms408/experiments/e4_root_leaf.py +282 -0
- ms408/experiments/e4b_reannotate.py +263 -0
- ms408/experiments/e5_encoding_fair.py +565 -0
- ms408/experiments/e6_cipher_reconstruction.py +338 -0
- ms408/experiments/e7_fine_anchor.py +305 -0
- ms408/experiments/e8_whitened_bracket.py +332 -0
- ms408/experiments/e9_vms_coordinate.py +204 -0
- ms408/experiments/mid_level_null.py +42 -0
- ms408/h4.py +270 -0
- ms408/harness/__init__.py +1 -0
- ms408/harness/naibbe.py +303 -0
- ms408/harness/selfcitation.py +1221 -0
- ms408/ivtff.py +184 -0
- ms408/mz.py +107 -0
- ms408/replication.py +798 -0
- ms408/scanmap.py +84 -0
- ms408/scans.py +123 -0
- ms408/signature.py +257 -0
- ms408/sources.py +151 -0
- ms408/studies/__init__.py +1 -0
- ms408/studies/anchor_hunt.py +363 -0
- ms408/studies/anchor_labels.py +276 -0
- ms408/studies/encoding.py +335 -0
- ms408/studies/morphology.py +373 -0
- ms408/studies/referential_realism.py +331 -0
- ms408/studies/topics.py +374 -0
- ms408/studies/variants.py +428 -0
- ms408/synthesis/__init__.py +5 -0
- ms408/synthesis/narratives.py +259 -0
- ms408/synthesis/registry.py +213 -0
- ms408/textstats.py +145 -0
- ms408/verify.py +109 -0
- ms408-0.1.0.dist-info/METADATA +188 -0
- ms408-0.1.0.dist-info/RECORD +83 -0
- ms408-0.1.0.dist-info/WHEEL +4 -0
- ms408-0.1.0.dist-info/entry_points.txt +2 -0
- ms408-0.1.0.dist-info/licenses/LICENSE +202 -0
- ms408-0.1.0.dist-info/licenses/NOTICE +21 -0
ms408/__init__.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""MS408 research program: corpus pipeline, validation harness, statistics.
|
|
2
|
+
|
|
3
|
+
All reported numbers must come from scripts in this package writing to results/ (L3 firewall).
|
|
4
|
+
|
|
5
|
+
Public evaluator entry point:
|
|
6
|
+
|
|
7
|
+
from ms408 import evaluate
|
|
8
|
+
verdict = evaluate(open("my_tokens.txt").read().split())
|
|
9
|
+
|
|
10
|
+
`evaluate(tokens)` scores a word-token stream against the Voynich manuscript's
|
|
11
|
+
discriminator bands and returns a per-axis verdict with each axis's honest caveat
|
|
12
|
+
attached. Matching is necessary, not sufficient (L7). See `ms408.signature` and
|
|
13
|
+
docs/LIMITS.md.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from .signature import axis_values, evaluate, format_verdict, vms_bands
|
|
17
|
+
|
|
18
|
+
__all__ = ["evaluate", "axis_values", "vms_bands", "format_verdict"]
|
ms408/__main__.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""CLI: evaluate a token stream against the VMS discriminator bands.
|
|
2
|
+
|
|
3
|
+
python -m ms408 my_tokens.txt # whitespace-separated word tokens
|
|
4
|
+
python -m ms408 --json my_tokens.txt # machine-readable verdict
|
|
5
|
+
cat my_tokens.txt | python -m ms408 - # read tokens from stdin
|
|
6
|
+
|
|
7
|
+
The verdict carries each axis's value, the VMS reference band, whether you land in it, and
|
|
8
|
+
the standing caveat for that axis. Matching is NECESSARY, not sufficient (L7): an in-band
|
|
9
|
+
result means your hypothesis is not excluded, not that it is the manuscript's mechanism.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import argparse
|
|
15
|
+
import json
|
|
16
|
+
import sys
|
|
17
|
+
|
|
18
|
+
from .signature import evaluate, format_verdict
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def main(argv: list | None = None) -> int:
|
|
22
|
+
ap = argparse.ArgumentParser(
|
|
23
|
+
prog="ms408",
|
|
24
|
+
description="Evaluate a token stream against the Voynich discriminator bands.",
|
|
25
|
+
)
|
|
26
|
+
ap.add_argument("tokens", help="path to a whitespace-separated token file, or '-' for stdin")
|
|
27
|
+
ap.add_argument("--json", action="store_true", help="emit the raw verdict as JSON")
|
|
28
|
+
ap.add_argument("--seed", type=int, default=408, help="determinism seed (default 408)")
|
|
29
|
+
args = ap.parse_args(argv)
|
|
30
|
+
|
|
31
|
+
text = sys.stdin.read() if args.tokens == "-" else open(args.tokens, encoding="utf-8").read()
|
|
32
|
+
tokens = text.split()
|
|
33
|
+
if len(tokens) < 2:
|
|
34
|
+
ap.error("need at least 2 whitespace-separated tokens")
|
|
35
|
+
|
|
36
|
+
verdict = evaluate(tokens, seed=args.seed)
|
|
37
|
+
if args.json:
|
|
38
|
+
print(json.dumps(verdict, indent=2))
|
|
39
|
+
else:
|
|
40
|
+
print(format_verdict(verdict))
|
|
41
|
+
return 0
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
if __name__ == "__main__":
|
|
45
|
+
raise SystemExit(main())
|
ms408/acquire.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Fetch and verify pinned external sources (T0.2).
|
|
2
|
+
|
|
3
|
+
Usage:
|
|
4
|
+
python -m ms408.acquire # fetch anything missing, verify everything
|
|
5
|
+
python -m ms408.acquire zl gc # fetch/verify named sources only
|
|
6
|
+
python -m ms408.acquire --verify # verify only, no network
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
import sys
|
|
13
|
+
import time
|
|
14
|
+
import urllib.request
|
|
15
|
+
|
|
16
|
+
from .sources import RAW_ROOT, SOURCES, Source, path_for
|
|
17
|
+
|
|
18
|
+
FETCH_DELAY_S = 1.0 # politeness between requests to the same host
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def sha256_of(path) -> str:
|
|
22
|
+
h = hashlib.sha256()
|
|
23
|
+
with open(path, "rb") as f:
|
|
24
|
+
for chunk in iter(lambda: f.read(1 << 20), b""):
|
|
25
|
+
h.update(chunk)
|
|
26
|
+
return h.hexdigest()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def fetch(source: Source, force: bool = False) -> str:
|
|
30
|
+
"""Download one source if missing (or force), then verify. Returns status."""
|
|
31
|
+
dest = RAW_ROOT / source.dest
|
|
32
|
+
if dest.exists() and not force:
|
|
33
|
+
return verify(source)
|
|
34
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
35
|
+
req = urllib.request.Request(source.url, headers={"User-Agent": "ms408-research/0.1"})
|
|
36
|
+
with urllib.request.urlopen(req) as resp, open(dest, "wb") as out:
|
|
37
|
+
out.write(resp.read())
|
|
38
|
+
time.sleep(FETCH_DELAY_S)
|
|
39
|
+
return verify(source)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def verify(source: Source) -> str:
|
|
43
|
+
dest = RAW_ROOT / source.dest
|
|
44
|
+
if not dest.exists():
|
|
45
|
+
return "missing"
|
|
46
|
+
return "ok" if sha256_of(dest) == source.sha256 else "CHECKSUM MISMATCH"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def main(argv: list[str] | None = None) -> int:
|
|
50
|
+
args = list(sys.argv[1:] if argv is None else argv)
|
|
51
|
+
verify_only = "--verify" in args
|
|
52
|
+
names = [a for a in args if not a.startswith("--")] or list(SOURCES)
|
|
53
|
+
failed = False
|
|
54
|
+
for name in names:
|
|
55
|
+
source = SOURCES[name]
|
|
56
|
+
status = verify(source) if verify_only else fetch(source)
|
|
57
|
+
print(f"{name:14s} {status:18s} {path_for(name)}")
|
|
58
|
+
if status != "ok":
|
|
59
|
+
failed = True
|
|
60
|
+
return 1 if failed else 0
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
if __name__ == "__main__":
|
|
64
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""T1.3 bulk illustration annotation (schema v0.1-coarse, L31).
|
|
2
|
+
|
|
3
|
+
Firewall note (L3/L1): the model produces descriptive feature values; it does
|
|
4
|
+
not compute any program statistic. Annotations feed the anchor hunt (T2.3),
|
|
5
|
+
which computes co-occurrence statistics in deterministic code.
|
|
6
|
+
"""
|
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
"""T1.3 bulk annotation runner (schema v0.1-coarse; budget cap L28 = $100).
|
|
2
|
+
|
|
3
|
+
Sonnet 4.6 does vision annotation via strict tool use (guaranteed schema-valid
|
|
4
|
+
output — replaces the Haiku format-validation step, which strict mode makes
|
|
5
|
+
redundant). Fable 5 QA is a separate module (qa.py). Cost is tracked per call
|
|
6
|
+
and the run aborts before exceeding the cap.
|
|
7
|
+
|
|
8
|
+
Resumable: pages already in the output JSONL are skipped, so a re-run continues.
|
|
9
|
+
|
|
10
|
+
Usage:
|
|
11
|
+
python -m ms408.annotate.pipeline --limit 5 # pilot
|
|
12
|
+
python -m ms408.annotate.pipeline --section H # one section
|
|
13
|
+
python -m ms408.annotate.pipeline # all illustrated pages
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import argparse
|
|
19
|
+
import base64
|
|
20
|
+
import json
|
|
21
|
+
from datetime import UTC, datetime
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
|
|
24
|
+
from ..dataset import git_commit
|
|
25
|
+
from ..env import require
|
|
26
|
+
from ..ivtff import IVTFFDocument
|
|
27
|
+
from ..scans import SCANS_ROOT
|
|
28
|
+
from ..sources import path_for
|
|
29
|
+
from .schema import SCHEMA_VERSION, SECTION_BLOCKS, tool_schema
|
|
30
|
+
|
|
31
|
+
ROOT = Path(__file__).resolve().parents[3]
|
|
32
|
+
RESULTS_DIR = ROOT / "results" / "annotations"
|
|
33
|
+
OUTPUT = RESULTS_DIR / "t13_annotations.jsonl"
|
|
34
|
+
MANIFEST = RESULTS_DIR / "t13_manifest.json"
|
|
35
|
+
|
|
36
|
+
ANNOTATOR_MODEL = "claude-sonnet-4-6"
|
|
37
|
+
BUDGET_CAP_USD = 100.0
|
|
38
|
+
# Sonnet 4.6 pricing, $/token
|
|
39
|
+
PRICE_IN = 3.0 / 1_000_000
|
|
40
|
+
PRICE_OUT = 15.0 / 1_000_000
|
|
41
|
+
|
|
42
|
+
SYSTEM = (
|
|
43
|
+
"You are a careful manuscript-illustration annotator for a scholarly project on "
|
|
44
|
+
"Beinecke MS 408 (the Voynich Manuscript). You describe what is drawn using a fixed "
|
|
45
|
+
"controlled vocabulary. Strict rules:\n"
|
|
46
|
+
"- Describe MORPHOLOGY only. Never identify a plant species, a zodiac sign, or a "
|
|
47
|
+
"real-world referent. 'A branched root', never 'a mandrake'. 'An animal', never 'a bull'.\n"
|
|
48
|
+
"- Judge only from what is visibly drawn. When the scan does not support a confident "
|
|
49
|
+
"call, choose 'unclear' rather than guessing.\n"
|
|
50
|
+
"- The manuscript's pigments are faded; read colors conservatively.\n"
|
|
51
|
+
"- Fill every field of the annotate_page tool. Put any ambiguity in 'notes' (brief, "
|
|
52
|
+
"descriptive, never identificational)."
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def illustrated_pages(section: str | None = None) -> list:
|
|
57
|
+
zl = IVTFFDocument.load(path_for("zl"))
|
|
58
|
+
pages = []
|
|
59
|
+
for page in zl.pages:
|
|
60
|
+
code = page.illustration_type
|
|
61
|
+
if code is None or code not in SECTION_BLOCKS:
|
|
62
|
+
continue
|
|
63
|
+
if section and code != section:
|
|
64
|
+
continue
|
|
65
|
+
pages.append((page.name, code))
|
|
66
|
+
return pages
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
MAX_EDGE_PX = 7000 # API rejects any image dimension > 8000px
|
|
70
|
+
# API's 10 MB cap is on the base64-ENCODED size; base64 inflates ~4/3, so keep
|
|
71
|
+
# the raw JPEG under ~7.3 MB (encoded ~9.8 MB, safely under 10,485,760)
|
|
72
|
+
MAX_RAW_BYTES = 7_300_000
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _image_block(path: Path) -> dict:
|
|
76
|
+
raw = path.read_bytes()
|
|
77
|
+
if max(_dimensions(path)) <= MAX_EDGE_PX and len(raw) <= MAX_RAW_BYTES:
|
|
78
|
+
data = base64.standard_b64encode(raw).decode()
|
|
79
|
+
return {"type": "image",
|
|
80
|
+
"source": {"type": "base64", "media_type": "image/jpeg", "data": data}}
|
|
81
|
+
|
|
82
|
+
import io
|
|
83
|
+
|
|
84
|
+
from PIL import Image
|
|
85
|
+
|
|
86
|
+
with Image.open(path) as img:
|
|
87
|
+
img = img.convert("RGB")
|
|
88
|
+
# cap the long edge, then shrink further if still over the byte budget
|
|
89
|
+
scale = min(1.0, MAX_EDGE_PX / max(img.size))
|
|
90
|
+
for _ in range(8):
|
|
91
|
+
w, h = round(img.width * scale), round(img.height * scale)
|
|
92
|
+
buffer = io.BytesIO()
|
|
93
|
+
img.resize((w, h), Image.LANCZOS).save(buffer, format="JPEG", quality=88)
|
|
94
|
+
raw = buffer.getvalue()
|
|
95
|
+
if len(raw) <= MAX_RAW_BYTES:
|
|
96
|
+
break
|
|
97
|
+
scale *= 0.85
|
|
98
|
+
data = base64.standard_b64encode(raw).decode()
|
|
99
|
+
return {"type": "image",
|
|
100
|
+
"source": {"type": "base64", "media_type": "image/jpeg", "data": data}}
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _dimensions(path: Path) -> tuple:
|
|
104
|
+
from PIL import Image
|
|
105
|
+
|
|
106
|
+
with Image.open(path) as img:
|
|
107
|
+
return img.size
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _load_scan_map() -> dict:
|
|
111
|
+
return json.loads((ROOT / "data" / "processed" / "scan_map.json").read_text())["pages"]
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _done_pages() -> set:
|
|
115
|
+
if not OUTPUT.exists():
|
|
116
|
+
return set()
|
|
117
|
+
return {json.loads(line)["page"] for line in OUTPUT.read_text().splitlines() if line.strip()}
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def annotate_page(client, page_name: str, code: str, scan_map: dict) -> dict:
|
|
121
|
+
entry = scan_map.get(page_name, {})
|
|
122
|
+
files = entry.get("files", [])
|
|
123
|
+
tool = {
|
|
124
|
+
"name": "annotate_page",
|
|
125
|
+
"description": "Record the coarse descriptive annotation for this manuscript page.",
|
|
126
|
+
"strict": True,
|
|
127
|
+
"input_schema": tool_schema(code),
|
|
128
|
+
}
|
|
129
|
+
content = [_image_block(SCANS_ROOT / f) for f in files]
|
|
130
|
+
section_name = SECTION_BLOCKS[code][0] or "text/stars (common block only)"
|
|
131
|
+
foldout_note = (
|
|
132
|
+
"Multiple scan tiles are provided for this foldout panel; annotate the panel "
|
|
133
|
+
"as a whole. " if len(files) > 1 else ""
|
|
134
|
+
)
|
|
135
|
+
content.append({"type": "text", "text":
|
|
136
|
+
f"This is folio {page_name}, illustration section '{section_name}'. "
|
|
137
|
+
f"{foldout_note}Call annotate_page with your descriptive annotation."})
|
|
138
|
+
response = client.messages.create(
|
|
139
|
+
model=ANNOTATOR_MODEL,
|
|
140
|
+
max_tokens=2000,
|
|
141
|
+
system=SYSTEM,
|
|
142
|
+
tools=[tool],
|
|
143
|
+
tool_choice={"type": "tool", "name": "annotate_page"},
|
|
144
|
+
messages=[{"role": "user", "content": content}],
|
|
145
|
+
)
|
|
146
|
+
features = next(b.input for b in response.content if b.type == "tool_use")
|
|
147
|
+
cost = (response.usage.input_tokens * PRICE_IN
|
|
148
|
+
+ response.usage.output_tokens * PRICE_OUT)
|
|
149
|
+
record = {
|
|
150
|
+
"schema_version": SCHEMA_VERSION,
|
|
151
|
+
"page": page_name,
|
|
152
|
+
"section": code,
|
|
153
|
+
"scan_tiles": files,
|
|
154
|
+
"foldout_ambiguous": entry.get("ambiguous", False),
|
|
155
|
+
"annotator_model": ANNOTATOR_MODEL,
|
|
156
|
+
"common": {k: features[k] for k in features if k in _COMMON_KEYS},
|
|
157
|
+
"section_features": {k: v for k, v in features.items()
|
|
158
|
+
if k not in _COMMON_KEYS and k != "notes"},
|
|
159
|
+
"notes": features.get("notes", ""),
|
|
160
|
+
"qa": {"reviewed": False, "reviewer_model": None, "disagreements": []},
|
|
161
|
+
"_cost_usd": round(cost, 5),
|
|
162
|
+
"_input_tokens": response.usage.input_tokens,
|
|
163
|
+
"_output_tokens": response.usage.output_tokens,
|
|
164
|
+
}
|
|
165
|
+
return record
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
_COMMON_KEYS = {"illustration_coverage_pct", "text_image_relationship", "color_palette"}
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def run(section: str | None = None, limit: int | None = None) -> dict:
|
|
172
|
+
import anthropic
|
|
173
|
+
|
|
174
|
+
require("ANTHROPIC_API_KEY")
|
|
175
|
+
client = anthropic.Anthropic()
|
|
176
|
+
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
|
|
177
|
+
scan_map = _load_scan_map()
|
|
178
|
+
done = _done_pages()
|
|
179
|
+
pages = [(n, c) for n, c in illustrated_pages(section) if n not in done]
|
|
180
|
+
if limit:
|
|
181
|
+
pages = pages[:limit]
|
|
182
|
+
|
|
183
|
+
spent = sum(json.loads(line).get("_cost_usd", 0)
|
|
184
|
+
for line in OUTPUT.read_text().splitlines()) if OUTPUT.exists() else 0.0
|
|
185
|
+
annotated = 0
|
|
186
|
+
with open(OUTPUT, "a") as out:
|
|
187
|
+
for page_name, code in pages:
|
|
188
|
+
if spent > BUDGET_CAP_USD:
|
|
189
|
+
print(f"BUDGET CAP ${BUDGET_CAP_USD} reached (${spent:.2f}); stopping.")
|
|
190
|
+
break
|
|
191
|
+
record = annotate_page(client, page_name, code, scan_map)
|
|
192
|
+
out.write(json.dumps(record) + "\n")
|
|
193
|
+
out.flush()
|
|
194
|
+
spent += record["_cost_usd"]
|
|
195
|
+
annotated += 1
|
|
196
|
+
print(f"{page_name:8s} {code} ${record['_cost_usd']:.4f} "
|
|
197
|
+
f"(total ${spent:.3f})")
|
|
198
|
+
|
|
199
|
+
manifest = {
|
|
200
|
+
"schema_version": SCHEMA_VERSION,
|
|
201
|
+
"built_at": datetime.now(UTC).isoformat(timespec="seconds"),
|
|
202
|
+
"git_commit": git_commit(),
|
|
203
|
+
"annotator_model": ANNOTATOR_MODEL,
|
|
204
|
+
"budget_cap_usd": BUDGET_CAP_USD,
|
|
205
|
+
"pages_annotated_total": len(_done_pages()),
|
|
206
|
+
"spent_usd_total": round(spent, 4),
|
|
207
|
+
}
|
|
208
|
+
MANIFEST.write_text(json.dumps(manifest, indent=2) + "\n")
|
|
209
|
+
return manifest
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def main(argv=None) -> int:
|
|
213
|
+
parser = argparse.ArgumentParser(description="T1.3 bulk annotation")
|
|
214
|
+
parser.add_argument("--section", choices=sorted(SECTION_BLOCKS))
|
|
215
|
+
parser.add_argument("--limit", type=int)
|
|
216
|
+
args = parser.parse_args(argv)
|
|
217
|
+
manifest = run(section=args.section, limit=args.limit)
|
|
218
|
+
print(json.dumps(manifest, indent=2))
|
|
219
|
+
return 0
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
if __name__ == "__main__":
|
|
223
|
+
raise SystemExit(main())
|
ms408/annotate/qa.py
ADDED
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""T1.3 QA — Fable 5 independent re-annotation of a sample, drift scoring (L31).
|
|
2
|
+
|
|
3
|
+
Per WORKFLOW §5: Fable 5 re-annotates a random 12% sample (min 5/batch, fixed
|
|
4
|
+
seed) blind; agreement is scored per field with the schema-derived metric
|
|
5
|
+
(exact for enum/bool/count-band, Jaccard≥0.67 for multi, ±1 for counts). Batch
|
|
6
|
+
fails if overall disagreement > 0.20, critical-field disagreement > 0.15, or any
|
|
7
|
+
single field disagrees on > 40% of sampled pages (L31 provisional thresholds).
|
|
8
|
+
|
|
9
|
+
Fable 5 never sees the Sonnet annotation — it produces an independent one from
|
|
10
|
+
the same scans, so agreement measures reproducibility, not self-consistency.
|
|
11
|
+
|
|
12
|
+
Usage:
|
|
13
|
+
python -m ms408.annotate.qa
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import json
|
|
19
|
+
import random
|
|
20
|
+
from datetime import UTC, datetime
|
|
21
|
+
|
|
22
|
+
from ..dataset import git_commit
|
|
23
|
+
from ..env import require
|
|
24
|
+
from ..scans import SCANS_ROOT
|
|
25
|
+
from .pipeline import (
|
|
26
|
+
OUTPUT,
|
|
27
|
+
RESULTS_DIR,
|
|
28
|
+
SYSTEM,
|
|
29
|
+
_COMMON_KEYS,
|
|
30
|
+
_image_block,
|
|
31
|
+
)
|
|
32
|
+
from .schema import SECTION_BLOCKS, critical_fields, fields_for, tool_schema
|
|
33
|
+
|
|
34
|
+
QA_MODEL = "claude-fable-5"
|
|
35
|
+
QA_FALLBACK = "claude-opus-4-8" # Fable false-positive-refuses manuscript annotation
|
|
36
|
+
QA_SAMPLE_FRACTION = 0.12
|
|
37
|
+
QA_SAMPLE_MIN = 5
|
|
38
|
+
SEED = 408
|
|
39
|
+
QA_OUTPUT = RESULTS_DIR / "t13_qa.json"
|
|
40
|
+
|
|
41
|
+
# ratified at G2 (2026-07-06, L32): the provisional 0.20/0.15 bands assumed
|
|
42
|
+
# same-model QA; these reflect real cross-model (Sonnet vs Opus/Fable) variance
|
|
43
|
+
# on coarse morphological calls. root_type/leaf_arrangement are known-noisy
|
|
44
|
+
# (~0.35, irreducible perceptual ambiguity) and pass single-field at 0.40.
|
|
45
|
+
THRESH_OVERALL = 0.25
|
|
46
|
+
THRESH_CRITICAL = 0.25
|
|
47
|
+
THRESH_SINGLE_FIELD = 0.40
|
|
48
|
+
|
|
49
|
+
# billed at whichever model served the call; fallback credit reprices refusals
|
|
50
|
+
PRICE = {"claude-fable-5": (10.0e-6, 50.0e-6),
|
|
51
|
+
"claude-opus-4-8": (5.0e-6, 25.0e-6)}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _agree(kind: str, a, b) -> bool:
|
|
55
|
+
if kind == "multi":
|
|
56
|
+
sa, sb = set(a or []), set(b or [])
|
|
57
|
+
if not sa and not sb:
|
|
58
|
+
return True
|
|
59
|
+
union = sa | sb
|
|
60
|
+
# threshold is exactly 2/3 (the schema's "0.67" rounding), so a 2-of-3
|
|
61
|
+
# overlap counts as agreement rather than failing by a rounding hair
|
|
62
|
+
return len(sa & sb) / len(union) >= 2 / 3 if union else True
|
|
63
|
+
if a is None or b is None:
|
|
64
|
+
return a == b # a genuinely absent field counts as a disagreement
|
|
65
|
+
if kind == "count":
|
|
66
|
+
return abs(int(a) - int(b)) <= 1
|
|
67
|
+
return a == b # enum, bool, count-band (enum strings)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def score_page(section: str, sonnet: dict, fable: dict) -> dict:
|
|
71
|
+
fields = fields_for(section)
|
|
72
|
+
critical = set(critical_fields(section))
|
|
73
|
+
disagreements = []
|
|
74
|
+
critical_disagreements = []
|
|
75
|
+
merged_a = {**sonnet["common"], **sonnet["section_features"]}
|
|
76
|
+
merged_b = {**fable["common"], **fable["section_features"]}
|
|
77
|
+
for field in fields:
|
|
78
|
+
if not _agree(field.kind, merged_a.get(field.name), merged_b.get(field.name)):
|
|
79
|
+
disagreements.append(field.name)
|
|
80
|
+
if field.name in critical:
|
|
81
|
+
critical_disagreements.append(field.name)
|
|
82
|
+
return {
|
|
83
|
+
"scored_fields": len(fields),
|
|
84
|
+
"disagreements": disagreements,
|
|
85
|
+
"critical_disagreements": critical_disagreements,
|
|
86
|
+
"page_disagreement_rate": round(len(disagreements) / len(fields), 4),
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def reannotate(client, record: dict) -> tuple:
|
|
91
|
+
code = record["section"]
|
|
92
|
+
tool = {"name": "annotate_page", "description": "Record the coarse descriptive "
|
|
93
|
+
"annotation for this manuscript page.", "strict": True,
|
|
94
|
+
"input_schema": tool_schema(code)}
|
|
95
|
+
content = [_image_block(SCANS_ROOT / f) for f in record["scan_tiles"]]
|
|
96
|
+
section_name = SECTION_BLOCKS[code][0] or "text/stars (common block only)"
|
|
97
|
+
content.append({"type": "text", "text":
|
|
98
|
+
f"This is folio {record['page']}, illustration section "
|
|
99
|
+
f"'{section_name}'. Call annotate_page with your descriptive annotation."})
|
|
100
|
+
# Fable 5 false-positive-refuses manuscript annotation; server-side fallback
|
|
101
|
+
# transparently re-serves the refusal on Opus 4.8 (an independent reviewer)
|
|
102
|
+
response = client.beta.messages.create(
|
|
103
|
+
model=QA_MODEL, max_tokens=3000, system=SYSTEM, tools=[tool],
|
|
104
|
+
tool_choice={"type": "tool", "name": "annotate_page"},
|
|
105
|
+
betas=["server-side-fallback-2026-06-01"],
|
|
106
|
+
fallbacks=[{"model": QA_FALLBACK}],
|
|
107
|
+
messages=[{"role": "user", "content": content}],
|
|
108
|
+
)
|
|
109
|
+
served_by = response.model
|
|
110
|
+
if response.stop_reason == "refusal":
|
|
111
|
+
return None, 0.0, served_by # whole chain refused; page skipped in QA
|
|
112
|
+
features = next((b.input for b in response.content if b.type == "tool_use"), {})
|
|
113
|
+
fable = {
|
|
114
|
+
"common": {k: features[k] for k in features if k in _COMMON_KEYS},
|
|
115
|
+
"section_features": {k: v for k, v in features.items()
|
|
116
|
+
if k not in _COMMON_KEYS and k != "notes"},
|
|
117
|
+
}
|
|
118
|
+
price_in, price_out = PRICE.get(served_by, PRICE[QA_FALLBACK])
|
|
119
|
+
cost = response.usage.input_tokens * price_in + response.usage.output_tokens * price_out
|
|
120
|
+
return fable, cost, served_by
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def run() -> dict:
|
|
124
|
+
import anthropic
|
|
125
|
+
from collections import Counter
|
|
126
|
+
|
|
127
|
+
require("ANTHROPIC_API_KEY")
|
|
128
|
+
client = anthropic.Anthropic()
|
|
129
|
+
records = [json.loads(line) for line in OUTPUT.read_text().splitlines() if line.strip()]
|
|
130
|
+
|
|
131
|
+
# stratified sample: at least QA_SAMPLE_MIN, ~12%, seeded, spread across sections
|
|
132
|
+
rng = random.Random(SEED)
|
|
133
|
+
by_section: dict = {}
|
|
134
|
+
for r in records:
|
|
135
|
+
by_section.setdefault(r["section"], []).append(r)
|
|
136
|
+
sample = []
|
|
137
|
+
for section, group in sorted(by_section.items()):
|
|
138
|
+
k = max(1, round(len(group) * QA_SAMPLE_FRACTION))
|
|
139
|
+
sample.extend(rng.sample(group, min(k, len(group))))
|
|
140
|
+
if len(sample) < QA_SAMPLE_MIN:
|
|
141
|
+
remaining = [r for r in records if r not in sample]
|
|
142
|
+
sample.extend(rng.sample(remaining, min(QA_SAMPLE_MIN - len(sample), len(remaining))))
|
|
143
|
+
|
|
144
|
+
scored, field_disagreements, spent = [], Counter(), 0.0
|
|
145
|
+
served_by = Counter()
|
|
146
|
+
refused = []
|
|
147
|
+
for record in sample:
|
|
148
|
+
fable, cost, model = reannotate(client, record)
|
|
149
|
+
spent += cost
|
|
150
|
+
served_by[model] += 1
|
|
151
|
+
if fable is None:
|
|
152
|
+
refused.append(record["page"])
|
|
153
|
+
continue
|
|
154
|
+
result = score_page(record["section"], record, fable)
|
|
155
|
+
result.update({"page": record["page"], "section": record["section"],
|
|
156
|
+
"qa_served_by": model})
|
|
157
|
+
field_disagreements.update(result["disagreements"])
|
|
158
|
+
scored.append(result)
|
|
159
|
+
print(f"{record['page']:8s} {record['section']} {model:16s} "
|
|
160
|
+
f"disagree={result['page_disagreement_rate']:.2f} "
|
|
161
|
+
f"critical={len(result['critical_disagreements'])}")
|
|
162
|
+
|
|
163
|
+
total_fields = sum(s["scored_fields"] for s in scored)
|
|
164
|
+
total_disagree = sum(len(s["disagreements"]) for s in scored)
|
|
165
|
+
total_critical_fields = sum(len(critical_fields(s["section"])) for s in scored)
|
|
166
|
+
total_critical_disagree = sum(len(s["critical_disagreements"]) for s in scored)
|
|
167
|
+
worst_field, worst_count = (field_disagreements.most_common(1) or [(None, 0)])[0]
|
|
168
|
+
|
|
169
|
+
overall = total_disagree / total_fields if total_fields else 0.0
|
|
170
|
+
critical_rate = (total_critical_disagree / total_critical_fields
|
|
171
|
+
if total_critical_fields else 0.0)
|
|
172
|
+
worst_single = worst_count / len(sample) if sample else 0.0
|
|
173
|
+
passed = (overall <= THRESH_OVERALL and critical_rate <= THRESH_CRITICAL
|
|
174
|
+
and worst_single <= THRESH_SINGLE_FIELD)
|
|
175
|
+
|
|
176
|
+
report = {
|
|
177
|
+
"built_at": datetime.now(UTC).isoformat(timespec="seconds"),
|
|
178
|
+
"git_commit": git_commit(),
|
|
179
|
+
"qa_model": QA_MODEL,
|
|
180
|
+
"qa_served_by": dict(served_by),
|
|
181
|
+
"qa_refused_pages": refused,
|
|
182
|
+
"sampled_pages": len(sample),
|
|
183
|
+
"scored_pages": len(scored),
|
|
184
|
+
"sample_seed": SEED,
|
|
185
|
+
"overall_disagreement_rate": round(overall, 4),
|
|
186
|
+
"critical_disagreement_rate": round(critical_rate, 4),
|
|
187
|
+
"worst_field": worst_field,
|
|
188
|
+
"worst_field_rate": round(worst_single, 4),
|
|
189
|
+
"thresholds": {"overall": THRESH_OVERALL, "critical": THRESH_CRITICAL,
|
|
190
|
+
"single_field": THRESH_SINGLE_FIELD},
|
|
191
|
+
"batch_passed": passed,
|
|
192
|
+
"field_disagreement_counts": dict(field_disagreements.most_common()),
|
|
193
|
+
"per_page": scored,
|
|
194
|
+
"qa_cost_usd": round(spent, 4),
|
|
195
|
+
}
|
|
196
|
+
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
|
|
197
|
+
QA_OUTPUT.write_text(json.dumps(report, indent=2) + "\n")
|
|
198
|
+
return report
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
if __name__ == "__main__":
|
|
202
|
+
report = run()
|
|
203
|
+
print(json.dumps({k: v for k, v in report.items()
|
|
204
|
+
if k not in ("per_page", "field_disagreement_counts")}, indent=2))
|