ms408 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. ms408/__init__.py +18 -0
  2. ms408/__main__.py +45 -0
  3. ms408/acquire.py +64 -0
  4. ms408/annotate/__init__.py +6 -0
  5. ms408/annotate/pipeline.py +223 -0
  6. ms408/annotate/qa.py +204 -0
  7. ms408/annotate/schema.py +217 -0
  8. ms408/benchmark.py +245 -0
  9. ms408/data/__init__.py +7 -0
  10. ms408/data/reference_bands.json +122 -0
  11. ms408/dataset.py +97 -0
  12. ms408/env.py +46 -0
  13. ms408/experiments/__init__.py +5 -0
  14. ms408/experiments/e10_third_rater.py +312 -0
  15. ms408/experiments/e11_style_control.py +184 -0
  16. ms408/experiments/e12_independent_rater.py +446 -0
  17. ms408/experiments/e13_function_content.py +290 -0
  18. ms408/experiments/e13b_function_content.py +181 -0
  19. ms408/experiments/e13c_function_content_corrected.py +156 -0
  20. ms408/experiments/e13d_band_robustness.py +169 -0
  21. ms408/experiments/e14_word_classes.py +210 -0
  22. ms408/experiments/e14b_word_class_robustness.py +166 -0
  23. ms408/experiments/e15_morphology.py +206 -0
  24. ms408/experiments/e15b_morphology_corrected.py +167 -0
  25. ms408/experiments/e17_ab_contrast.py +168 -0
  26. ms408/experiments/e18_corpus_completeness.py +107 -0
  27. ms408/experiments/e19_joint_signature.py +187 -0
  28. ms408/experiments/e19b_language_universality.py +139 -0
  29. ms408/experiments/e1_meaning_detector.py +253 -0
  30. ms408/experiments/e20_transposition_closure.py +148 -0
  31. ms408/experiments/e21_positional_generator.py +408 -0
  32. ms408/experiments/e22_generator_genericity.py +273 -0
  33. ms408/experiments/e23_reuse_generator.py +294 -0
  34. ms408/experiments/e24_typelevel_lexicon.py +287 -0
  35. ms408/experiments/e25_decoupled_ed1.py +359 -0
  36. ms408/experiments/e26_length_variance.py +277 -0
  37. ms408/experiments/e27_symbol_quantification.py +228 -0
  38. ms408/experiments/e28_angular_anchor.py +350 -0
  39. ms408/experiments/e29_naibbe_discriminators.py +265 -0
  40. ms408/experiments/e2_wordorder_confound.py +282 -0
  41. ms408/experiments/e30_cipher_reexamination.py +223 -0
  42. ms408/experiments/e31_harden_syntax.py +264 -0
  43. ms408/experiments/e32_reference_bands.py +158 -0
  44. ms408/experiments/e33_block_scale_di.py +281 -0
  45. ms408/experiments/e3_anchor_power.py +269 -0
  46. ms408/experiments/e4_root_leaf.py +282 -0
  47. ms408/experiments/e4b_reannotate.py +263 -0
  48. ms408/experiments/e5_encoding_fair.py +565 -0
  49. ms408/experiments/e6_cipher_reconstruction.py +338 -0
  50. ms408/experiments/e7_fine_anchor.py +305 -0
  51. ms408/experiments/e8_whitened_bracket.py +332 -0
  52. ms408/experiments/e9_vms_coordinate.py +204 -0
  53. ms408/experiments/mid_level_null.py +42 -0
  54. ms408/h4.py +270 -0
  55. ms408/harness/__init__.py +1 -0
  56. ms408/harness/naibbe.py +303 -0
  57. ms408/harness/selfcitation.py +1221 -0
  58. ms408/ivtff.py +184 -0
  59. ms408/mz.py +107 -0
  60. ms408/replication.py +798 -0
  61. ms408/scanmap.py +84 -0
  62. ms408/scans.py +123 -0
  63. ms408/signature.py +257 -0
  64. ms408/sources.py +151 -0
  65. ms408/studies/__init__.py +1 -0
  66. ms408/studies/anchor_hunt.py +363 -0
  67. ms408/studies/anchor_labels.py +276 -0
  68. ms408/studies/encoding.py +335 -0
  69. ms408/studies/morphology.py +373 -0
  70. ms408/studies/referential_realism.py +331 -0
  71. ms408/studies/topics.py +374 -0
  72. ms408/studies/variants.py +428 -0
  73. ms408/synthesis/__init__.py +5 -0
  74. ms408/synthesis/narratives.py +259 -0
  75. ms408/synthesis/registry.py +213 -0
  76. ms408/textstats.py +145 -0
  77. ms408/verify.py +109 -0
  78. ms408-0.1.0.dist-info/METADATA +188 -0
  79. ms408-0.1.0.dist-info/RECORD +83 -0
  80. ms408-0.1.0.dist-info/WHEEL +4 -0
  81. ms408-0.1.0.dist-info/entry_points.txt +2 -0
  82. ms408-0.1.0.dist-info/licenses/LICENSE +202 -0
  83. ms408-0.1.0.dist-info/licenses/NOTICE +21 -0
ms408/__init__.py ADDED
@@ -0,0 +1,18 @@
1
+ """MS408 research program: corpus pipeline, validation harness, statistics.
2
+
3
+ All reported numbers must come from scripts in this package writing to results/ (L3 firewall).
4
+
5
+ Public evaluator entry point:
6
+
7
+ from ms408 import evaluate
8
+ verdict = evaluate(open("my_tokens.txt").read().split())
9
+
10
+ `evaluate(tokens)` scores a word-token stream against the Voynich manuscript's
11
+ discriminator bands and returns a per-axis verdict with each axis's honest caveat
12
+ attached. Matching is necessary, not sufficient (L7). See `ms408.signature` and
13
+ docs/LIMITS.md.
14
+ """
15
+
16
+ from .signature import axis_values, evaluate, format_verdict, vms_bands
17
+
18
+ __all__ = ["evaluate", "axis_values", "vms_bands", "format_verdict"]
ms408/__main__.py ADDED
@@ -0,0 +1,45 @@
1
+ """CLI: evaluate a token stream against the VMS discriminator bands.
2
+
3
+ python -m ms408 my_tokens.txt # whitespace-separated word tokens
4
+ python -m ms408 --json my_tokens.txt # machine-readable verdict
5
+ cat my_tokens.txt | python -m ms408 - # read tokens from stdin
6
+
7
+ The verdict carries each axis's value, the VMS reference band, whether you land in it, and
8
+ the standing caveat for that axis. Matching is NECESSARY, not sufficient (L7): an in-band
9
+ result means your hypothesis is not excluded, not that it is the manuscript's mechanism.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import argparse
15
+ import json
16
+ import sys
17
+
18
+ from .signature import evaluate, format_verdict
19
+
20
+
21
+ def main(argv: list | None = None) -> int:
22
+ ap = argparse.ArgumentParser(
23
+ prog="ms408",
24
+ description="Evaluate a token stream against the Voynich discriminator bands.",
25
+ )
26
+ ap.add_argument("tokens", help="path to a whitespace-separated token file, or '-' for stdin")
27
+ ap.add_argument("--json", action="store_true", help="emit the raw verdict as JSON")
28
+ ap.add_argument("--seed", type=int, default=408, help="determinism seed (default 408)")
29
+ args = ap.parse_args(argv)
30
+
31
+ text = sys.stdin.read() if args.tokens == "-" else open(args.tokens, encoding="utf-8").read()
32
+ tokens = text.split()
33
+ if len(tokens) < 2:
34
+ ap.error("need at least 2 whitespace-separated tokens")
35
+
36
+ verdict = evaluate(tokens, seed=args.seed)
37
+ if args.json:
38
+ print(json.dumps(verdict, indent=2))
39
+ else:
40
+ print(format_verdict(verdict))
41
+ return 0
42
+
43
+
44
+ if __name__ == "__main__":
45
+ raise SystemExit(main())
ms408/acquire.py ADDED
@@ -0,0 +1,64 @@
1
+ """Fetch and verify pinned external sources (T0.2).
2
+
3
+ Usage:
4
+ python -m ms408.acquire # fetch anything missing, verify everything
5
+ python -m ms408.acquire zl gc # fetch/verify named sources only
6
+ python -m ms408.acquire --verify # verify only, no network
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ import sys
13
+ import time
14
+ import urllib.request
15
+
16
+ from .sources import RAW_ROOT, SOURCES, Source, path_for
17
+
18
+ FETCH_DELAY_S = 1.0 # politeness between requests to the same host
19
+
20
+
21
+ def sha256_of(path) -> str:
22
+ h = hashlib.sha256()
23
+ with open(path, "rb") as f:
24
+ for chunk in iter(lambda: f.read(1 << 20), b""):
25
+ h.update(chunk)
26
+ return h.hexdigest()
27
+
28
+
29
+ def fetch(source: Source, force: bool = False) -> str:
30
+ """Download one source if missing (or force), then verify. Returns status."""
31
+ dest = RAW_ROOT / source.dest
32
+ if dest.exists() and not force:
33
+ return verify(source)
34
+ dest.parent.mkdir(parents=True, exist_ok=True)
35
+ req = urllib.request.Request(source.url, headers={"User-Agent": "ms408-research/0.1"})
36
+ with urllib.request.urlopen(req) as resp, open(dest, "wb") as out:
37
+ out.write(resp.read())
38
+ time.sleep(FETCH_DELAY_S)
39
+ return verify(source)
40
+
41
+
42
+ def verify(source: Source) -> str:
43
+ dest = RAW_ROOT / source.dest
44
+ if not dest.exists():
45
+ return "missing"
46
+ return "ok" if sha256_of(dest) == source.sha256 else "CHECKSUM MISMATCH"
47
+
48
+
49
+ def main(argv: list[str] | None = None) -> int:
50
+ args = list(sys.argv[1:] if argv is None else argv)
51
+ verify_only = "--verify" in args
52
+ names = [a for a in args if not a.startswith("--")] or list(SOURCES)
53
+ failed = False
54
+ for name in names:
55
+ source = SOURCES[name]
56
+ status = verify(source) if verify_only else fetch(source)
57
+ print(f"{name:14s} {status:18s} {path_for(name)}")
58
+ if status != "ok":
59
+ failed = True
60
+ return 1 if failed else 0
61
+
62
+
63
+ if __name__ == "__main__":
64
+ raise SystemExit(main())
@@ -0,0 +1,6 @@
1
+ """T1.3 bulk illustration annotation (schema v0.1-coarse, L31).
2
+
3
+ Firewall note (L3/L1): the model produces descriptive feature values; it does
4
+ not compute any program statistic. Annotations feed the anchor hunt (T2.3),
5
+ which computes co-occurrence statistics in deterministic code.
6
+ """
@@ -0,0 +1,223 @@
1
+ """T1.3 bulk annotation runner (schema v0.1-coarse; budget cap L28 = $100).
2
+
3
+ Sonnet 4.6 does vision annotation via strict tool use (guaranteed schema-valid
4
+ output — replaces the Haiku format-validation step, which strict mode makes
5
+ redundant). Fable 5 QA is a separate module (qa.py). Cost is tracked per call
6
+ and the run aborts before exceeding the cap.
7
+
8
+ Resumable: pages already in the output JSONL are skipped, so a re-run continues.
9
+
10
+ Usage:
11
+ python -m ms408.annotate.pipeline --limit 5 # pilot
12
+ python -m ms408.annotate.pipeline --section H # one section
13
+ python -m ms408.annotate.pipeline # all illustrated pages
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import argparse
19
+ import base64
20
+ import json
21
+ from datetime import UTC, datetime
22
+ from pathlib import Path
23
+
24
+ from ..dataset import git_commit
25
+ from ..env import require
26
+ from ..ivtff import IVTFFDocument
27
+ from ..scans import SCANS_ROOT
28
+ from ..sources import path_for
29
+ from .schema import SCHEMA_VERSION, SECTION_BLOCKS, tool_schema
30
+
31
+ ROOT = Path(__file__).resolve().parents[3]
32
+ RESULTS_DIR = ROOT / "results" / "annotations"
33
+ OUTPUT = RESULTS_DIR / "t13_annotations.jsonl"
34
+ MANIFEST = RESULTS_DIR / "t13_manifest.json"
35
+
36
+ ANNOTATOR_MODEL = "claude-sonnet-4-6"
37
+ BUDGET_CAP_USD = 100.0
38
+ # Sonnet 4.6 pricing, $/token
39
+ PRICE_IN = 3.0 / 1_000_000
40
+ PRICE_OUT = 15.0 / 1_000_000
41
+
42
+ SYSTEM = (
43
+ "You are a careful manuscript-illustration annotator for a scholarly project on "
44
+ "Beinecke MS 408 (the Voynich Manuscript). You describe what is drawn using a fixed "
45
+ "controlled vocabulary. Strict rules:\n"
46
+ "- Describe MORPHOLOGY only. Never identify a plant species, a zodiac sign, or a "
47
+ "real-world referent. 'A branched root', never 'a mandrake'. 'An animal', never 'a bull'.\n"
48
+ "- Judge only from what is visibly drawn. When the scan does not support a confident "
49
+ "call, choose 'unclear' rather than guessing.\n"
50
+ "- The manuscript's pigments are faded; read colors conservatively.\n"
51
+ "- Fill every field of the annotate_page tool. Put any ambiguity in 'notes' (brief, "
52
+ "descriptive, never identificational)."
53
+ )
54
+
55
+
56
+ def illustrated_pages(section: str | None = None) -> list:
57
+ zl = IVTFFDocument.load(path_for("zl"))
58
+ pages = []
59
+ for page in zl.pages:
60
+ code = page.illustration_type
61
+ if code is None or code not in SECTION_BLOCKS:
62
+ continue
63
+ if section and code != section:
64
+ continue
65
+ pages.append((page.name, code))
66
+ return pages
67
+
68
+
69
+ MAX_EDGE_PX = 7000 # API rejects any image dimension > 8000px
70
+ # API's 10 MB cap is on the base64-ENCODED size; base64 inflates ~4/3, so keep
71
+ # the raw JPEG under ~7.3 MB (encoded ~9.8 MB, safely under 10,485,760)
72
+ MAX_RAW_BYTES = 7_300_000
73
+
74
+
75
+ def _image_block(path: Path) -> dict:
76
+ raw = path.read_bytes()
77
+ if max(_dimensions(path)) <= MAX_EDGE_PX and len(raw) <= MAX_RAW_BYTES:
78
+ data = base64.standard_b64encode(raw).decode()
79
+ return {"type": "image",
80
+ "source": {"type": "base64", "media_type": "image/jpeg", "data": data}}
81
+
82
+ import io
83
+
84
+ from PIL import Image
85
+
86
+ with Image.open(path) as img:
87
+ img = img.convert("RGB")
88
+ # cap the long edge, then shrink further if still over the byte budget
89
+ scale = min(1.0, MAX_EDGE_PX / max(img.size))
90
+ for _ in range(8):
91
+ w, h = round(img.width * scale), round(img.height * scale)
92
+ buffer = io.BytesIO()
93
+ img.resize((w, h), Image.LANCZOS).save(buffer, format="JPEG", quality=88)
94
+ raw = buffer.getvalue()
95
+ if len(raw) <= MAX_RAW_BYTES:
96
+ break
97
+ scale *= 0.85
98
+ data = base64.standard_b64encode(raw).decode()
99
+ return {"type": "image",
100
+ "source": {"type": "base64", "media_type": "image/jpeg", "data": data}}
101
+
102
+
103
+ def _dimensions(path: Path) -> tuple:
104
+ from PIL import Image
105
+
106
+ with Image.open(path) as img:
107
+ return img.size
108
+
109
+
110
+ def _load_scan_map() -> dict:
111
+ return json.loads((ROOT / "data" / "processed" / "scan_map.json").read_text())["pages"]
112
+
113
+
114
+ def _done_pages() -> set:
115
+ if not OUTPUT.exists():
116
+ return set()
117
+ return {json.loads(line)["page"] for line in OUTPUT.read_text().splitlines() if line.strip()}
118
+
119
+
120
+ def annotate_page(client, page_name: str, code: str, scan_map: dict) -> dict:
121
+ entry = scan_map.get(page_name, {})
122
+ files = entry.get("files", [])
123
+ tool = {
124
+ "name": "annotate_page",
125
+ "description": "Record the coarse descriptive annotation for this manuscript page.",
126
+ "strict": True,
127
+ "input_schema": tool_schema(code),
128
+ }
129
+ content = [_image_block(SCANS_ROOT / f) for f in files]
130
+ section_name = SECTION_BLOCKS[code][0] or "text/stars (common block only)"
131
+ foldout_note = (
132
+ "Multiple scan tiles are provided for this foldout panel; annotate the panel "
133
+ "as a whole. " if len(files) > 1 else ""
134
+ )
135
+ content.append({"type": "text", "text":
136
+ f"This is folio {page_name}, illustration section '{section_name}'. "
137
+ f"{foldout_note}Call annotate_page with your descriptive annotation."})
138
+ response = client.messages.create(
139
+ model=ANNOTATOR_MODEL,
140
+ max_tokens=2000,
141
+ system=SYSTEM,
142
+ tools=[tool],
143
+ tool_choice={"type": "tool", "name": "annotate_page"},
144
+ messages=[{"role": "user", "content": content}],
145
+ )
146
+ features = next(b.input for b in response.content if b.type == "tool_use")
147
+ cost = (response.usage.input_tokens * PRICE_IN
148
+ + response.usage.output_tokens * PRICE_OUT)
149
+ record = {
150
+ "schema_version": SCHEMA_VERSION,
151
+ "page": page_name,
152
+ "section": code,
153
+ "scan_tiles": files,
154
+ "foldout_ambiguous": entry.get("ambiguous", False),
155
+ "annotator_model": ANNOTATOR_MODEL,
156
+ "common": {k: features[k] for k in features if k in _COMMON_KEYS},
157
+ "section_features": {k: v for k, v in features.items()
158
+ if k not in _COMMON_KEYS and k != "notes"},
159
+ "notes": features.get("notes", ""),
160
+ "qa": {"reviewed": False, "reviewer_model": None, "disagreements": []},
161
+ "_cost_usd": round(cost, 5),
162
+ "_input_tokens": response.usage.input_tokens,
163
+ "_output_tokens": response.usage.output_tokens,
164
+ }
165
+ return record
166
+
167
+
168
+ _COMMON_KEYS = {"illustration_coverage_pct", "text_image_relationship", "color_palette"}
169
+
170
+
171
+ def run(section: str | None = None, limit: int | None = None) -> dict:
172
+ import anthropic
173
+
174
+ require("ANTHROPIC_API_KEY")
175
+ client = anthropic.Anthropic()
176
+ RESULTS_DIR.mkdir(parents=True, exist_ok=True)
177
+ scan_map = _load_scan_map()
178
+ done = _done_pages()
179
+ pages = [(n, c) for n, c in illustrated_pages(section) if n not in done]
180
+ if limit:
181
+ pages = pages[:limit]
182
+
183
+ spent = sum(json.loads(line).get("_cost_usd", 0)
184
+ for line in OUTPUT.read_text().splitlines()) if OUTPUT.exists() else 0.0
185
+ annotated = 0
186
+ with open(OUTPUT, "a") as out:
187
+ for page_name, code in pages:
188
+ if spent > BUDGET_CAP_USD:
189
+ print(f"BUDGET CAP ${BUDGET_CAP_USD} reached (${spent:.2f}); stopping.")
190
+ break
191
+ record = annotate_page(client, page_name, code, scan_map)
192
+ out.write(json.dumps(record) + "\n")
193
+ out.flush()
194
+ spent += record["_cost_usd"]
195
+ annotated += 1
196
+ print(f"{page_name:8s} {code} ${record['_cost_usd']:.4f} "
197
+ f"(total ${spent:.3f})")
198
+
199
+ manifest = {
200
+ "schema_version": SCHEMA_VERSION,
201
+ "built_at": datetime.now(UTC).isoformat(timespec="seconds"),
202
+ "git_commit": git_commit(),
203
+ "annotator_model": ANNOTATOR_MODEL,
204
+ "budget_cap_usd": BUDGET_CAP_USD,
205
+ "pages_annotated_total": len(_done_pages()),
206
+ "spent_usd_total": round(spent, 4),
207
+ }
208
+ MANIFEST.write_text(json.dumps(manifest, indent=2) + "\n")
209
+ return manifest
210
+
211
+
212
+ def main(argv=None) -> int:
213
+ parser = argparse.ArgumentParser(description="T1.3 bulk annotation")
214
+ parser.add_argument("--section", choices=sorted(SECTION_BLOCKS))
215
+ parser.add_argument("--limit", type=int)
216
+ args = parser.parse_args(argv)
217
+ manifest = run(section=args.section, limit=args.limit)
218
+ print(json.dumps(manifest, indent=2))
219
+ return 0
220
+
221
+
222
+ if __name__ == "__main__":
223
+ raise SystemExit(main())
ms408/annotate/qa.py ADDED
@@ -0,0 +1,204 @@
1
+ """T1.3 QA — Fable 5 independent re-annotation of a sample, drift scoring (L31).
2
+
3
+ Per WORKFLOW §5: Fable 5 re-annotates a random 12% sample (min 5/batch, fixed
4
+ seed) blind; agreement is scored per field with the schema-derived metric
5
+ (exact for enum/bool/count-band, Jaccard≥0.67 for multi, ±1 for counts). Batch
6
+ fails if overall disagreement > 0.20, critical-field disagreement > 0.15, or any
7
+ single field disagrees on > 40% of sampled pages (L31 provisional thresholds).
8
+
9
+ Fable 5 never sees the Sonnet annotation — it produces an independent one from
10
+ the same scans, so agreement measures reproducibility, not self-consistency.
11
+
12
+ Usage:
13
+ python -m ms408.annotate.qa
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import json
19
+ import random
20
+ from datetime import UTC, datetime
21
+
22
+ from ..dataset import git_commit
23
+ from ..env import require
24
+ from ..scans import SCANS_ROOT
25
+ from .pipeline import (
26
+ OUTPUT,
27
+ RESULTS_DIR,
28
+ SYSTEM,
29
+ _COMMON_KEYS,
30
+ _image_block,
31
+ )
32
+ from .schema import SECTION_BLOCKS, critical_fields, fields_for, tool_schema
33
+
34
+ QA_MODEL = "claude-fable-5"
35
+ QA_FALLBACK = "claude-opus-4-8" # Fable false-positive-refuses manuscript annotation
36
+ QA_SAMPLE_FRACTION = 0.12
37
+ QA_SAMPLE_MIN = 5
38
+ SEED = 408
39
+ QA_OUTPUT = RESULTS_DIR / "t13_qa.json"
40
+
41
+ # ratified at G2 (2026-07-06, L32): the provisional 0.20/0.15 bands assumed
42
+ # same-model QA; these reflect real cross-model (Sonnet vs Opus/Fable) variance
43
+ # on coarse morphological calls. root_type/leaf_arrangement are known-noisy
44
+ # (~0.35, irreducible perceptual ambiguity) and pass single-field at 0.40.
45
+ THRESH_OVERALL = 0.25
46
+ THRESH_CRITICAL = 0.25
47
+ THRESH_SINGLE_FIELD = 0.40
48
+
49
+ # billed at whichever model served the call; fallback credit reprices refusals
50
+ PRICE = {"claude-fable-5": (10.0e-6, 50.0e-6),
51
+ "claude-opus-4-8": (5.0e-6, 25.0e-6)}
52
+
53
+
54
+ def _agree(kind: str, a, b) -> bool:
55
+ if kind == "multi":
56
+ sa, sb = set(a or []), set(b or [])
57
+ if not sa and not sb:
58
+ return True
59
+ union = sa | sb
60
+ # threshold is exactly 2/3 (the schema's "0.67" rounding), so a 2-of-3
61
+ # overlap counts as agreement rather than failing by a rounding hair
62
+ return len(sa & sb) / len(union) >= 2 / 3 if union else True
63
+ if a is None or b is None:
64
+ return a == b # a genuinely absent field counts as a disagreement
65
+ if kind == "count":
66
+ return abs(int(a) - int(b)) <= 1
67
+ return a == b # enum, bool, count-band (enum strings)
68
+
69
+
70
+ def score_page(section: str, sonnet: dict, fable: dict) -> dict:
71
+ fields = fields_for(section)
72
+ critical = set(critical_fields(section))
73
+ disagreements = []
74
+ critical_disagreements = []
75
+ merged_a = {**sonnet["common"], **sonnet["section_features"]}
76
+ merged_b = {**fable["common"], **fable["section_features"]}
77
+ for field in fields:
78
+ if not _agree(field.kind, merged_a.get(field.name), merged_b.get(field.name)):
79
+ disagreements.append(field.name)
80
+ if field.name in critical:
81
+ critical_disagreements.append(field.name)
82
+ return {
83
+ "scored_fields": len(fields),
84
+ "disagreements": disagreements,
85
+ "critical_disagreements": critical_disagreements,
86
+ "page_disagreement_rate": round(len(disagreements) / len(fields), 4),
87
+ }
88
+
89
+
90
+ def reannotate(client, record: dict) -> tuple:
91
+ code = record["section"]
92
+ tool = {"name": "annotate_page", "description": "Record the coarse descriptive "
93
+ "annotation for this manuscript page.", "strict": True,
94
+ "input_schema": tool_schema(code)}
95
+ content = [_image_block(SCANS_ROOT / f) for f in record["scan_tiles"]]
96
+ section_name = SECTION_BLOCKS[code][0] or "text/stars (common block only)"
97
+ content.append({"type": "text", "text":
98
+ f"This is folio {record['page']}, illustration section "
99
+ f"'{section_name}'. Call annotate_page with your descriptive annotation."})
100
+ # Fable 5 false-positive-refuses manuscript annotation; server-side fallback
101
+ # transparently re-serves the refusal on Opus 4.8 (an independent reviewer)
102
+ response = client.beta.messages.create(
103
+ model=QA_MODEL, max_tokens=3000, system=SYSTEM, tools=[tool],
104
+ tool_choice={"type": "tool", "name": "annotate_page"},
105
+ betas=["server-side-fallback-2026-06-01"],
106
+ fallbacks=[{"model": QA_FALLBACK}],
107
+ messages=[{"role": "user", "content": content}],
108
+ )
109
+ served_by = response.model
110
+ if response.stop_reason == "refusal":
111
+ return None, 0.0, served_by # whole chain refused; page skipped in QA
112
+ features = next((b.input for b in response.content if b.type == "tool_use"), {})
113
+ fable = {
114
+ "common": {k: features[k] for k in features if k in _COMMON_KEYS},
115
+ "section_features": {k: v for k, v in features.items()
116
+ if k not in _COMMON_KEYS and k != "notes"},
117
+ }
118
+ price_in, price_out = PRICE.get(served_by, PRICE[QA_FALLBACK])
119
+ cost = response.usage.input_tokens * price_in + response.usage.output_tokens * price_out
120
+ return fable, cost, served_by
121
+
122
+
123
+ def run() -> dict:
124
+ import anthropic
125
+ from collections import Counter
126
+
127
+ require("ANTHROPIC_API_KEY")
128
+ client = anthropic.Anthropic()
129
+ records = [json.loads(line) for line in OUTPUT.read_text().splitlines() if line.strip()]
130
+
131
+ # stratified sample: at least QA_SAMPLE_MIN, ~12%, seeded, spread across sections
132
+ rng = random.Random(SEED)
133
+ by_section: dict = {}
134
+ for r in records:
135
+ by_section.setdefault(r["section"], []).append(r)
136
+ sample = []
137
+ for section, group in sorted(by_section.items()):
138
+ k = max(1, round(len(group) * QA_SAMPLE_FRACTION))
139
+ sample.extend(rng.sample(group, min(k, len(group))))
140
+ if len(sample) < QA_SAMPLE_MIN:
141
+ remaining = [r for r in records if r not in sample]
142
+ sample.extend(rng.sample(remaining, min(QA_SAMPLE_MIN - len(sample), len(remaining))))
143
+
144
+ scored, field_disagreements, spent = [], Counter(), 0.0
145
+ served_by = Counter()
146
+ refused = []
147
+ for record in sample:
148
+ fable, cost, model = reannotate(client, record)
149
+ spent += cost
150
+ served_by[model] += 1
151
+ if fable is None:
152
+ refused.append(record["page"])
153
+ continue
154
+ result = score_page(record["section"], record, fable)
155
+ result.update({"page": record["page"], "section": record["section"],
156
+ "qa_served_by": model})
157
+ field_disagreements.update(result["disagreements"])
158
+ scored.append(result)
159
+ print(f"{record['page']:8s} {record['section']} {model:16s} "
160
+ f"disagree={result['page_disagreement_rate']:.2f} "
161
+ f"critical={len(result['critical_disagreements'])}")
162
+
163
+ total_fields = sum(s["scored_fields"] for s in scored)
164
+ total_disagree = sum(len(s["disagreements"]) for s in scored)
165
+ total_critical_fields = sum(len(critical_fields(s["section"])) for s in scored)
166
+ total_critical_disagree = sum(len(s["critical_disagreements"]) for s in scored)
167
+ worst_field, worst_count = (field_disagreements.most_common(1) or [(None, 0)])[0]
168
+
169
+ overall = total_disagree / total_fields if total_fields else 0.0
170
+ critical_rate = (total_critical_disagree / total_critical_fields
171
+ if total_critical_fields else 0.0)
172
+ worst_single = worst_count / len(sample) if sample else 0.0
173
+ passed = (overall <= THRESH_OVERALL and critical_rate <= THRESH_CRITICAL
174
+ and worst_single <= THRESH_SINGLE_FIELD)
175
+
176
+ report = {
177
+ "built_at": datetime.now(UTC).isoformat(timespec="seconds"),
178
+ "git_commit": git_commit(),
179
+ "qa_model": QA_MODEL,
180
+ "qa_served_by": dict(served_by),
181
+ "qa_refused_pages": refused,
182
+ "sampled_pages": len(sample),
183
+ "scored_pages": len(scored),
184
+ "sample_seed": SEED,
185
+ "overall_disagreement_rate": round(overall, 4),
186
+ "critical_disagreement_rate": round(critical_rate, 4),
187
+ "worst_field": worst_field,
188
+ "worst_field_rate": round(worst_single, 4),
189
+ "thresholds": {"overall": THRESH_OVERALL, "critical": THRESH_CRITICAL,
190
+ "single_field": THRESH_SINGLE_FIELD},
191
+ "batch_passed": passed,
192
+ "field_disagreement_counts": dict(field_disagreements.most_common()),
193
+ "per_page": scored,
194
+ "qa_cost_usd": round(spent, 4),
195
+ }
196
+ RESULTS_DIR.mkdir(parents=True, exist_ok=True)
197
+ QA_OUTPUT.write_text(json.dumps(report, indent=2) + "\n")
198
+ return report
199
+
200
+
201
+ if __name__ == "__main__":
202
+ report = run()
203
+ print(json.dumps({k: v for k, v in report.items()
204
+ if k not in ("per_page", "field_disagreement_counts")}, indent=2))