embedflow 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. embedflow/__init__.py +25 -0
  2. embedflow/__main__.py +3 -0
  3. embedflow/analysis.py +192 -0
  4. embedflow/cache/__init__.py +4 -0
  5. embedflow/cache/base.py +28 -0
  6. embedflow/cache/persistent_cache.py +198 -0
  7. embedflow/cli.py +1200 -0
  8. embedflow/compatibility/__init__.py +28 -0
  9. embedflow/compatibility/candidate_gap.py +105 -0
  10. embedflow/compatibility/containment.py +17 -0
  11. embedflow/compatibility/evaluate.py +319 -0
  12. embedflow/compatibility/metrics.py +75 -0
  13. embedflow/compatibility/migration_depth.py +67 -0
  14. embedflow/compatibility/probe.py +34 -0
  15. embedflow/compatibility/report.py +102 -0
  16. embedflow/compatibility/t2.py +64 -0
  17. embedflow/config.py +455 -0
  18. embedflow/data/__init__.py +1 -0
  19. embedflow/data/registry/__init__.py +1 -0
  20. embedflow/data/registry/benchmark_profiles.jsonl +3 -0
  21. embedflow/data/registry/checksums.sha256 +4 -0
  22. embedflow/data/registry/migrations.jsonl +15 -0
  23. embedflow/data/registry/registry_manifest.json +16 -0
  24. embedflow/data/registry/research_summaries.json +55 -0
  25. embedflow/data/registry/schema_version.json +5 -0
  26. embedflow/frozen/T2_V1_FROZEN_SPEC.md +71 -0
  27. embedflow/frozen/T2_V1_FROZEN_SPEC.sha256 +1 -0
  28. embedflow/indexes/__init__.py +5 -0
  29. embedflow/indexes/base.py +60 -0
  30. embedflow/indexes/faiss_backend.py +240 -0
  31. embedflow/indexes/qdrant_backend.py +225 -0
  32. embedflow/metrics/__init__.py +3 -0
  33. embedflow/metrics/latency.py +50 -0
  34. embedflow/migration/__init__.py +3 -0
  35. embedflow/migration/compatibility.py +156 -0
  36. embedflow/migration/facade.py +312 -0
  37. embedflow/migration/materializer.py +190 -0
  38. embedflow/migration/planner.py +78 -0
  39. embedflow/migration/state.py +81 -0
  40. embedflow/models/__init__.py +4 -0
  41. embedflow/models/base.py +31 -0
  42. embedflow/models/huggingface.py +226 -0
  43. embedflow/registry/__init__.py +47 -0
  44. embedflow/registry/loader.py +785 -0
  45. embedflow/registry/matcher.py +197 -0
  46. embedflow/registry/schema.py +266 -0
  47. embedflow/runtime.py +115 -0
  48. embedflow/serving/__init__.py +3 -0
  49. embedflow/serving/api.py +161 -0
  50. embedflow/serving/engine.py +222 -0
  51. embedflow/serving/factory.py +3 -0
  52. embedflow/serving/schemas.py +39 -0
  53. embedflow-0.1.0.dist-info/METADATA +210 -0
  54. embedflow-0.1.0.dist-info/RECORD +64 -0
  55. embedflow-0.1.0.dist-info/WHEEL +5 -0
  56. embedflow-0.1.0.dist-info/entry_points.txt +2 -0
  57. embedflow-0.1.0.dist-info/licenses/LICENSE +178 -0
  58. embedflow-0.1.0.dist-info/top_level.txt +2 -0
  59. src/__init__.py +1 -0
  60. src/embed.py +123 -0
  61. src/probe_features.py +24 -0
  62. src/storage.py +51 -0
  63. src/t2_v1.py +21 -0
  64. src/utils.py +53 -0
@@ -0,0 +1,785 @@
1
+ from __future__ import annotations
2
+
3
+ import csv
4
+ import hashlib
5
+ import importlib.resources as resources
6
+ import json
7
+ import math
8
+ from collections.abc import Iterable, Mapping
9
+ from pathlib import Path
10
+ from typing import Any
11
+
12
+ from .schema import REGISTRY_VERSION, SCHEMA_VERSION, BenchmarkProfile, EvidenceRecord, RegistryError
13
+
14
+
15
+ def _resource_root() -> Any:
16
+ return resources.files("embedflow.data.registry")
17
+
18
+
19
+ def _read_json(name: str, root: Any | None = None) -> dict[str, Any]:
20
+ resource = (root or _resource_root()).joinpath(name)
21
+ try:
22
+ return json.loads(resource.read_text(encoding="utf-8"))
23
+ except FileNotFoundError as exc:
24
+ raise RegistryError(f"registry file is missing: {name}") from exc
25
+ except json.JSONDecodeError as exc:
26
+ raise RegistryError(f"registry file is not valid JSON: {name}: {exc}") from exc
27
+
28
+
29
+ def _read_jsonl(name: str, root: Any | None = None) -> list[dict[str, Any]]:
30
+ resource = (root or _resource_root()).joinpath(name)
31
+ try:
32
+ text = resource.read_text(encoding="utf-8")
33
+ except FileNotFoundError as exc:
34
+ raise RegistryError(f"registry file is missing: {name}") from exc
35
+ rows: list[dict[str, Any]] = []
36
+ for line_number, line in enumerate(text.splitlines(), 1):
37
+ if not line.strip():
38
+ continue
39
+ try:
40
+ value = json.loads(line)
41
+ except json.JSONDecodeError as exc:
42
+ raise RegistryError(f"{name}:{line_number}: invalid JSON: {exc}") from exc
43
+ if not isinstance(value, dict):
44
+ raise RegistryError(f"{name}:{line_number}: each row must be a JSON object")
45
+ rows.append(value)
46
+ return rows
47
+
48
+
49
+ def load_manifest(path: str | Path | None = None) -> dict[str, Any]:
50
+ """Load the packaged registry manifest or a directory-level manifest."""
51
+ if path is None:
52
+ return _read_json("registry_manifest.json")
53
+ candidate = Path(path)
54
+ if candidate.is_dir():
55
+ candidate = candidate / "registry_manifest.json"
56
+ try:
57
+ payload = json.loads(candidate.read_text(encoding="utf-8"))
58
+ except (OSError, UnicodeError, json.JSONDecodeError) as exc:
59
+ raise RegistryError(f"registry manifest is not valid JSON: {candidate}") from exc
60
+ if not isinstance(payload, dict):
61
+ raise RegistryError(f"registry manifest must be a JSON object: {candidate}")
62
+ return payload
63
+
64
+
65
+ def _root_for(path: str | Path | None) -> Any | None:
66
+ if path is None:
67
+ return None
68
+ candidate = Path(path)
69
+ return candidate if candidate.is_dir() else candidate.parent
70
+
71
+
72
+ def load_evidence(path: str | Path | None = None) -> list[EvidenceRecord]:
73
+ """Load and validate deterministic migration records."""
74
+ root = _root_for(path)
75
+ name = "migrations.jsonl"
76
+ rows = _read_jsonl(name, root) if root is not None else _read_jsonl(name)
77
+ evidence = [EvidenceRecord.from_dict(row) for row in rows]
78
+ return sorted(evidence, key=lambda item: item.evidence_id)
79
+
80
+
81
+ def load_benchmark_profiles(path: str | Path | None = None) -> list[BenchmarkProfile]:
82
+ """Load measured latency/throughput profiles, separate from compatibility rows."""
83
+ root = _root_for(path)
84
+ name = "benchmark_profiles.jsonl"
85
+ rows = _read_jsonl(name, root) if root is not None else _read_jsonl(name)
86
+ profiles = [BenchmarkProfile.from_dict(row) for row in rows]
87
+ return sorted(profiles, key=lambda item: item.profile_id)
88
+
89
+
90
+ def load_summaries(path: str | Path | None = None) -> list[dict[str, Any]]:
91
+ root = _root_for(path)
92
+ name = "research_summaries.json"
93
+ payload = _read_json(name, root) if root is not None else _read_json(name)
94
+ summaries = payload.get("summaries", [])
95
+ if not isinstance(summaries, list) or not all(isinstance(row, dict) for row in summaries):
96
+ raise RegistryError("research_summaries.json field 'summaries' must be a list of objects")
97
+ return list(summaries)
98
+
99
+
100
+ def _sha256_bytes(value: bytes) -> str:
101
+ return hashlib.sha256(value).hexdigest()
102
+
103
+
104
+ def _resource_bytes(root: Any, name: str) -> bytes:
105
+ return root.joinpath(name).read_bytes()
106
+
107
+
108
+ def _verify_checksums(root: Any, manifest: Mapping[str, Any], errors: list[str]) -> int:
109
+ checked = 0
110
+ files = manifest.get("files") or {}
111
+ if not isinstance(files, Mapping):
112
+ errors.append("manifest field 'files' must be an object")
113
+ files = {}
114
+ for name, expected in files.items():
115
+ if not isinstance(expected, dict) or not expected.get("sha256"):
116
+ errors.append(f"manifest entry {name!r} lacks sha256")
117
+ continue
118
+ try:
119
+ actual = _sha256_bytes(_resource_bytes(root, name))
120
+ except FileNotFoundError:
121
+ errors.append(f"manifest file is missing: {name}")
122
+ continue
123
+ checked += 1
124
+ if actual != str(expected["sha256"]):
125
+ errors.append(f"checksum mismatch for {name}: expected {expected['sha256']}, got {actual}")
126
+ checksum_resource = root.joinpath("checksums.sha256")
127
+ if checksum_resource.is_file():
128
+ for line_number, line in enumerate(checksum_resource.read_text(encoding="utf-8").splitlines(), 1):
129
+ if not line.strip() or line.lstrip().startswith("#"):
130
+ continue
131
+ parts = line.split(maxsplit=1)
132
+ if len(parts) != 2 or len(parts[0]) != 64:
133
+ errors.append(f"checksums.sha256:{line_number}: malformed checksum line")
134
+ continue
135
+ name, expected = parts[1].lstrip(" *"), parts[0]
136
+ try:
137
+ actual = _sha256_bytes(_resource_bytes(root, name))
138
+ except FileNotFoundError:
139
+ errors.append(f"checksums.sha256:{line_number}: missing {name}")
140
+ continue
141
+ if actual != expected:
142
+ errors.append(f"checksums.sha256:{line_number}: checksum mismatch for {name}")
143
+ return checked
144
+
145
+
146
+ def _verify_external_provenance(
147
+ provenance_root: str | Path | None,
148
+ evidence: list[EvidenceRecord],
149
+ profiles: list[BenchmarkProfile],
150
+ summaries: list[dict[str, Any]],
151
+ errors: list[str],
152
+ warnings: list[str],
153
+ ) -> int:
154
+ """Check retained artifact bytes when the research checkout is present.
155
+
156
+ Installed users normally do not have the private/research checkout, so a
157
+ missing external artifact is a warning. A present artifact with a wrong
158
+ digest is a release-blocking error: packaged numbers must remain tied to
159
+ the exact retained file from which they were transcribed.
160
+ """
161
+ if provenance_root is None:
162
+ return 0
163
+ root = Path(provenance_root).expanduser().resolve()
164
+ checked = 0
165
+ rows: list[tuple[str, Mapping[str, Any]]] = []
166
+ rows.extend((row.evidence_id, row.raw.get("provenance", {})) for row in evidence)
167
+ rows.extend((profile.profile_id, profile.raw.get("provenance", {})) for profile in profiles)
168
+ rows.extend((str(row.get("summary_id", "summary")), row.get("provenance", {})) for row in summaries)
169
+ for identifier, provenance in rows:
170
+ if not isinstance(provenance, Mapping):
171
+ warnings.append(f"{identifier}: external artifact digest is unavailable")
172
+ continue
173
+ # A summary may name a primary row-level table and a supporting audit
174
+ # table (for example, the 63-cell result table plus its CI headline
175
+ # audit). Verify every declared digest rather than silently trusting
176
+ # only the first path.
177
+ artifacts = [("artifact", provenance.get("artifact"), provenance.get("artifact_sha256"))]
178
+ if provenance.get("supporting_artifact") or provenance.get("supporting_artifact_sha256"):
179
+ artifacts.append(("supporting_artifact", provenance.get("supporting_artifact"), provenance.get("supporting_artifact_sha256")))
180
+ for label, artifact, expected in artifacts:
181
+ if not artifact or not expected:
182
+ warnings.append(f"{identifier}: {label} digest is unavailable")
183
+ continue
184
+ candidate = Path(str(artifact))
185
+ if not candidate.is_absolute():
186
+ candidate = root / candidate
187
+ if not candidate.exists():
188
+ warnings.append(f"{identifier}: retained {label} not found at {candidate}")
189
+ continue
190
+ try:
191
+ actual = hashlib.sha256(candidate.read_bytes()).hexdigest()
192
+ except OSError as exc:
193
+ errors.append(f"{identifier}: cannot read retained {label} {candidate}: {exc}")
194
+ continue
195
+ checked += 1
196
+ if actual != str(expected):
197
+ errors.append(f"{identifier}: retained {label} checksum mismatch: expected {expected}, got {actual}")
198
+ return checked
199
+
200
+
201
+ def _artifact_path(provenance_root: Path, artifact: Any) -> Path:
202
+ """Resolve a provenance path relative to the research checkout root."""
203
+ candidate = Path(str(artifact)).expanduser()
204
+ return candidate if candidate.is_absolute() else provenance_root / candidate
205
+
206
+
207
+ def _as_float(value: Any, label: str) -> float:
208
+ try:
209
+ result = float(value)
210
+ except (TypeError, ValueError) as exc:
211
+ raise RegistryError(f"{label} is not numeric") from exc
212
+ if not math.isfinite(result):
213
+ raise RegistryError(f"{label} is not finite")
214
+ return result
215
+
216
+
217
+ def _same_measurement(actual: Any, expected: Any) -> bool:
218
+ """Compare retained decimal output without requiring byte-identical floats."""
219
+ try:
220
+ actual_value = _as_float(actual, "artifact value")
221
+ expected_value = _as_float(expected, "registry value")
222
+ except RegistryError:
223
+ return False
224
+ return math.isclose(actual_value, expected_value, rel_tol=1e-10, abs_tol=1e-12)
225
+
226
+
227
+ def _verify_curve_row(record: EvidenceRecord, row: Mapping[str, Any], *, source_fields: tuple[str, str], gap_field: str, containment_field: str, errors: list[str]) -> int:
228
+ """Verify one registry migration row against one retained curve row."""
229
+ checked = 0
230
+ try:
231
+ k = int(row["K"])
232
+ except (KeyError, TypeError, ValueError):
233
+ errors.append(f"{record.evidence_id}: retained curve row has an invalid K")
234
+ return checked
235
+ expected_gap = record.candidate_gap.get(k)
236
+ if expected_gap is None:
237
+ errors.append(f"{record.evidence_id}: registry has no candidate_gap for retained K={k}")
238
+ return checked
239
+ comparisons = (
240
+ (record.raw.get("source_quality"), source_fields[0], "source quality"),
241
+ (record.raw.get("native_target_quality"), source_fields[1], "native target quality"),
242
+ (record.raw.get("restricted_target_quality", {}).get(str(k)), "restricted_ndcg" if gap_field == "G" else "restricted_target_ndcg", "restricted target quality"),
243
+ (expected_gap, gap_field, "candidate gap"),
244
+ )
245
+ for expected, field, label in comparisons:
246
+ if expected is None or field not in row or not _same_measurement(row[field], expected):
247
+ errors.append(f"{record.evidence_id}: retained {label} disagrees at K={k}")
248
+ else:
249
+ checked += 1
250
+ expected_containment = record.containment.get(k)
251
+ if expected_containment is not None:
252
+ if containment_field not in row or not _same_measurement(row[containment_field], expected_containment):
253
+ errors.append(f"{record.evidence_id}: retained containment disagrees at K={k}")
254
+ else:
255
+ checked += 1
256
+ if record.dataset.get("query_count") is not None:
257
+ try:
258
+ if int(row.get("n_queries", row.get("queries"))) != int(record.dataset["query_count"]):
259
+ errors.append(f"{record.evidence_id}: retained query count disagrees")
260
+ else:
261
+ checked += 1
262
+ except (TypeError, ValueError):
263
+ errors.append(f"{record.evidence_id}: retained query count is invalid")
264
+ return checked
265
+
266
+
267
+ def _verify_semantic_provenance(
268
+ provenance_root: str | Path | None,
269
+ evidence: list[EvidenceRecord],
270
+ profiles: list[BenchmarkProfile],
271
+ summaries: list[dict[str, Any]],
272
+ errors: list[str],
273
+ warnings: list[str],
274
+ ) -> int:
275
+ """Cross-check transcribed registry values against known retained tables.
276
+
277
+ Checksums establish that an artifact was not changed; these checks establish
278
+ that the values in the public registry are actually the values in that
279
+ artifact. Unknown future artifact formats are left to contributors rather
280
+ than guessed here.
281
+ """
282
+ if provenance_root is None:
283
+ return 0
284
+ root = Path(provenance_root).expanduser().resolve()
285
+ checked = 0
286
+ by_artifact: dict[str, list[EvidenceRecord]] = {}
287
+ for record in evidence:
288
+ artifact = record.raw.get("provenance", {}).get("artifact")
289
+ if artifact:
290
+ by_artifact.setdefault(str(artifact), []).append(record)
291
+ for artifact, records in by_artifact.items():
292
+ path = _artifact_path(root, artifact)
293
+ if not path.is_file():
294
+ continue
295
+ try:
296
+ with path.open(newline="", encoding="utf-8") as handle:
297
+ rows = list(csv.DictReader(handle))
298
+ except (OSError, csv.Error) as exc:
299
+ errors.append(f"semantic provenance: cannot parse {path}: {exc}")
300
+ continue
301
+ name = path.name
302
+ if name == "scale_curves.csv":
303
+ pair_by_source = {
304
+ "sentence-transformers/all-MiniLM-L6-v2": "minilm_l6_to_qwen3_8b",
305
+ "Qwen/Qwen3-Embedding-0.6B": "qwen3_0_6b_to_qwen3_8b",
306
+ "Qwen/Qwen3-Embedding-4B": "qwen3_4b_to_qwen3_8b",
307
+ }
308
+ for record in records:
309
+ pair = pair_by_source.get(str(record.source.get("canonical_model_id")))
310
+ corpus_size = record.dataset.get("corpus_size")
311
+ if pair is None or corpus_size is None:
312
+ warnings.append(f"{record.evidence_id}: no semantic parser mapping for {name}")
313
+ continue
314
+ matches = [r for r in rows if r.get("pair") == pair and r.get("corpus_size") == str(corpus_size)]
315
+ if len(matches) != len(record.candidate_gap):
316
+ errors.append(f"{record.evidence_id}: retained scale curve has {len(matches)} rows; expected {len(record.candidate_gap)}")
317
+ continue
318
+ for row in matches:
319
+ checked += _verify_curve_row(record, row, source_fields=("source_ndcg", "native_target_ndcg"), gap_field="G", containment_field="containment10", errors=errors)
320
+ elif name == "bright_pooled_appendix_values.csv":
321
+ source_by_key = {
322
+ "sentence-transformers/all-MiniLM-L6-v2": "minilm_l6",
323
+ "Qwen/Qwen3-Embedding-0.6B": "qwen3_0_6b",
324
+ "Qwen/Qwen3-Embedding-4B": "qwen3_4b",
325
+ }
326
+ for record in records:
327
+ source_key = source_by_key.get(str(record.source.get("canonical_model_id")))
328
+ if source_key is None:
329
+ warnings.append(f"{record.evidence_id}: no semantic parser mapping for {name}")
330
+ continue
331
+ matches = [r for r in rows if r.get("source") == source_key]
332
+ if len(matches) != len(record.candidate_gap):
333
+ errors.append(f"{record.evidence_id}: retained BRIGHT curve has {len(matches)} rows; expected {len(record.candidate_gap)}")
334
+ continue
335
+ by_k: dict[int, Mapping[str, Any]] = {}
336
+ invalid_k = False
337
+ for retained in matches:
338
+ if not retained.get("K"):
339
+ invalid_k = True
340
+ continue
341
+ try:
342
+ parsed_k = int(retained["K"])
343
+ except (TypeError, ValueError):
344
+ invalid_k = True
345
+ continue
346
+ if parsed_k in by_k:
347
+ invalid_k = True
348
+ by_k[parsed_k] = retained
349
+ if invalid_k:
350
+ errors.append(f"{record.evidence_id}: retained BRIGHT curve contains an invalid or duplicate K")
351
+ if set(by_k) != set(record.candidate_gap):
352
+ errors.append(f"{record.evidence_id}: retained BRIGHT K grid disagrees")
353
+ continue
354
+ for row in matches:
355
+ checked += _verify_curve_row(record, row, source_fields=("source_ndcg", "target_ndcg"), gap_field="absolute_candidate_gap", containment_field="target_top10_containment", errors=errors)
356
+ # The BRIGHT appendix explicitly records the two depth notions;
357
+ # check both against the first qualifying K, preserving null.
358
+ point_depth = next((k for k, r in by_k.items() if str(r.get("observed_near_target_g_le_0_01", "")).lower() == "true"), None)
359
+ ci_depth = next((k for k, r in by_k.items() if str(r.get("target_noninferior_margin_0_01", "")).lower() == "true"), None)
360
+ if record.raw.get("observed_migration_depth") != point_depth:
361
+ errors.append(f"{record.evidence_id}: retained BRIGHT observed depth disagrees")
362
+ else:
363
+ checked += 1
364
+ if record.raw.get("ci_certified_migration_depth") != ci_depth:
365
+ errors.append(f"{record.evidence_id}: retained BRIGHT CI-certified depth disagrees")
366
+ else:
367
+ checked += 1
368
+ else:
369
+ warnings.append(f"semantic provenance: no parser for retained artifact {path}")
370
+ # Measured serving profiles are kept separate from migration evidence, but
371
+ # their headline values are checked against the retained latency/profile
372
+ # artifacts as well.
373
+ profile_by_artifact: dict[str, list[BenchmarkProfile]] = {}
374
+ for profile in profiles:
375
+ artifact = profile.raw.get("provenance", {}).get("artifact")
376
+ if artifact:
377
+ profile_by_artifact.setdefault(str(artifact), []).append(profile)
378
+ for artifact, profile_rows in profile_by_artifact.items():
379
+ path = _artifact_path(root, artifact)
380
+ if not path.is_file():
381
+ continue
382
+ if path.name == "latency_summary.csv":
383
+ try:
384
+ with path.open(newline="", encoding="utf-8") as handle:
385
+ latency_rows = list(csv.DictReader(handle))
386
+ except (OSError, csv.Error) as exc:
387
+ errors.append(f"semantic provenance: cannot parse {path}: {exc}")
388
+ continue
389
+ for profile in profile_rows:
390
+ config = profile.raw.get("configuration") or {}
391
+ if profile.profile_id.startswith("latency_25k_qwen3_4b"):
392
+ selected = [r for r in latency_rows if r.get("source_model") == "qwen3_4b" and r.get("mode") == "embedflow_warm" and r.get("K") == str(config.get("K")) and r.get("nprobe") == str(config.get("nprobe"))]
393
+ elif profile.profile_id.startswith("latency_25k_native"):
394
+ selected = [r for r in latency_rows if r.get("source_model") == "native_target" and r.get("mode") == "native_target"]
395
+ else:
396
+ warnings.append(f"{profile.profile_id}: no semantic parser mapping for {path.name}")
397
+ continue
398
+ if len(selected) != 1:
399
+ errors.append(f"{profile.profile_id}: retained latency selection has {len(selected)} rows")
400
+ continue
401
+ row = selected[0]
402
+ measurements = profile.raw.get("measurements") or {}
403
+ for key in ("p50_ms", "p95_ms"):
404
+ if key not in measurements or not _same_measurement(row.get(key), measurements[key]):
405
+ errors.append(f"{profile.profile_id}: retained {key} disagrees")
406
+ else:
407
+ checked += 1
408
+ if "added_vs_native_p50_ms" in measurements:
409
+ native = next((r for r in latency_rows if r.get("source_model") == "native_target" and r.get("mode") == "native_target"), None)
410
+ try:
411
+ actual_added = float(row["p50_ms"]) - float(native["p50_ms"]) if native is not None else math.nan
412
+ except (KeyError, TypeError, ValueError):
413
+ actual_added = math.nan
414
+ if not _same_measurement(actual_added, measurements["added_vs_native_p50_ms"]):
415
+ errors.append(f"{profile.profile_id}: retained added p50 disagrees")
416
+ else:
417
+ checked += 1
418
+ if "added_vs_native_p95_ms" in measurements:
419
+ native = next((r for r in latency_rows if r.get("source_model") == "native_target" and r.get("mode") == "native_target"), None)
420
+ try:
421
+ actual_added = float(row["p95_ms"]) - float(native["p95_ms"]) if native is not None else math.nan
422
+ except (KeyError, TypeError, ValueError):
423
+ actual_added = math.nan
424
+ if not _same_measurement(actual_added, measurements["added_vs_native_p95_ms"]):
425
+ errors.append(f"{profile.profile_id}: retained added p95 disagrees")
426
+ else:
427
+ checked += 1
428
+ for registry_name, csv_name in {
429
+ "source_query_encode": "mean_source_query_encode_ms",
430
+ "source_ann": "mean_source_ann_search_ms",
431
+ "target_query_encode": "mean_target_query_encode_ms",
432
+ "target_score": "mean_target_score_ms",
433
+ "topk": "mean_topk_ms",
434
+ }.items():
435
+ expected = (measurements.get("mean_stage_ms") or {}).get(registry_name)
436
+ if expected is not None:
437
+ if not _same_measurement(row.get(csv_name), expected):
438
+ errors.append(f"{profile.profile_id}: retained stage {registry_name} disagrees")
439
+ else:
440
+ checked += 1
441
+ expected = (measurements.get("mean_stage_ms") or {}).get("candidate_cache_lookup")
442
+ if expected is not None:
443
+ try:
444
+ actual = float(row.get("mean_candidate_lookup_ms", 0.0)) + float(row.get("mean_cache_lookup_ms", 0.0))
445
+ except (TypeError, ValueError):
446
+ actual = math.nan
447
+ if not _same_measurement(actual, expected):
448
+ errors.append(f"{profile.profile_id}: retained stage candidate_cache_lookup disagrees")
449
+ else:
450
+ checked += 1
451
+ elif path.name == "qwen3_8b_latency.json":
452
+ try:
453
+ payload = json.loads(path.read_text(encoding="utf-8"))
454
+ batch_rows = payload.get("batch_sizes", [])
455
+ except (OSError, json.JSONDecodeError, AttributeError) as exc:
456
+ errors.append(f"semantic provenance: cannot parse {path}: {exc}")
457
+ continue
458
+ if not isinstance(batch_rows, list) or not all(isinstance(row, Mapping) for row in batch_rows):
459
+ errors.append(f"semantic provenance: {path} batch_sizes must be a list of objects")
460
+ continue
461
+ for profile in profile_rows:
462
+ measurements = profile.raw.get("measurements") or {}
463
+ best_batch = measurements.get("best_batch_size")
464
+ selected = [r for r in batch_rows if r.get("batch_size") == best_batch]
465
+ if len(selected) != 1:
466
+ errors.append(f"{profile.profile_id}: retained throughput selection has {len(selected)} rows")
467
+ continue
468
+ row = selected[0]
469
+ for registry_name, artifact_name in (("best_rows_per_second", "rows_per_second"), ("best_mean_seconds", "mean_seconds")):
470
+ if not _same_measurement(row.get(artifact_name), measurements.get(registry_name)):
471
+ errors.append(f"{profile.profile_id}: retained {registry_name} disagrees")
472
+ else:
473
+ checked += 1
474
+ all_rows = measurements.get("all_batch_rows_per_second") or {}
475
+ for batch, expected in all_rows.items():
476
+ match = next((r for r in batch_rows if str(r.get("batch_size")) == str(batch)), None)
477
+ if match is None or not _same_measurement(match.get("rows_per_second"), expected):
478
+ errors.append(f"{profile.profile_id}: retained throughput disagrees for batch {batch}")
479
+ else:
480
+ checked += 1
481
+ else:
482
+ warnings.append(f"semantic provenance: no parser for retained profile artifact {path}")
483
+
484
+ # Summary values are count audits rather than curves. They are already
485
+ # byte-hash protected; verify the expected shape and headline counts when
486
+ # the canonical files are available so a swapped summary cannot pass
487
+ # unnoticed. The development audit has two retained representations: the
488
+ # 63-row result table carries G(50), while the appendix table carries the
489
+ # existing CI-certification flag.
490
+ for summary in summaries:
491
+ provenance = summary.get("provenance") or {}
492
+ declared_artifacts: list[tuple[str, Any]] = []
493
+ if provenance.get("artifact"):
494
+ declared_artifacts.append(("artifact", provenance.get("artifact")))
495
+ if provenance.get("supporting_artifact"):
496
+ declared_artifacts.append(("supporting_artifact", provenance.get("supporting_artifact")))
497
+ for artifact_label, artifact in declared_artifacts:
498
+ path = _artifact_path(root, artifact)
499
+ if not path.is_file() or path.suffix.lower() != ".csv":
500
+ continue
501
+ try:
502
+ with path.open(newline="", encoding="utf-8") as handle:
503
+ rows = list(csv.DictReader(handle))
504
+ except (OSError, csv.Error) as exc:
505
+ errors.append(f"semantic provenance: cannot parse {path}: {exc}")
506
+ continue
507
+ summary_id = str(summary.get("summary_id"))
508
+ if summary.get("kind") == "development_sweep_summary":
509
+ expected_cells = int(summary.get("cells", -1))
510
+ if len(rows) != expected_cells:
511
+ errors.append(f"{summary_id}: retained {artifact_label} development row count {len(rows)} != {expected_cells}")
512
+ continue
513
+ checked += 1
514
+ # Both the production result table and the appendix table
515
+ # encode the point-estimate G(50) under different names.
516
+ gap_field = "exact_top50_gap_ndcg" if rows and "exact_top50_gap_ndcg" in rows[0] else "signed_candidate_gap_g50"
517
+ if rows and gap_field in rows[0]:
518
+ try:
519
+ point_count = sum(_as_float(row.get(gap_field), f"{path} {gap_field}") <= 0.01 for row in rows)
520
+ except RegistryError as exc:
521
+ errors.append(f"{summary_id}: invalid development gap value in {path}: {exc}")
522
+ else:
523
+ declared_point = int(summary.get("point_estimate_g50_le_0_01", -1))
524
+ if point_count != declared_point:
525
+ errors.append(f"{summary_id}: retained point-estimate count {point_count} != {declared_point}")
526
+ else:
527
+ checked += 1
528
+ non_nano = [row for row in rows if str(row.get("dataset", "")).lower() not in {"nanoquora", "nanomsmarco"}]
529
+ expected_excluded = summary.get("excluding_nanoquora_and_nanomsmarco", {}).get("point_estimate_g50_le_0_01")
530
+ if expected_excluded is not None:
531
+ try:
532
+ excluded_count = sum(_as_float(row.get(gap_field), f"{path} {gap_field}") <= 0.01 for row in non_nano)
533
+ except RegistryError as exc:
534
+ errors.append(f"{summary_id}: invalid non-Nano development gap value in {path}: {exc}")
535
+ else:
536
+ if excluded_count != int(expected_excluded):
537
+ errors.append(f"{summary_id}: retained non-Nano point count {excluded_count} != {expected_excluded}")
538
+ else:
539
+ checked += 1
540
+ # Only the appendix representation contains the canonical
541
+ # existing CI flag. Do not substitute a different budget
542
+ # column from the development result table.
543
+ ci_field = "ci_noninferior_margin_0_01" if rows and "ci_noninferior_margin_0_01" in rows[0] else None
544
+ if ci_field:
545
+ ci_count = sum(str(row.get(ci_field, "")).strip().lower() == "true" for row in rows)
546
+ declared_ci = int(summary.get("ci_certified_cells", -1))
547
+ if ci_count != declared_ci:
548
+ errors.append(f"{summary_id}: retained CI-certified count {ci_count} != {declared_ci}")
549
+ else:
550
+ checked += 1
551
+ non_nano = [row for row in rows if str(row.get("dataset", "")).lower() not in {"nanoquora", "nanomsmarco"}]
552
+ expected_excluded_ci = summary.get("excluding_nanoquora_and_nanomsmarco", {}).get("ci_certified_cells")
553
+ if expected_excluded_ci is not None:
554
+ excluded_ci_count = sum(str(row.get(ci_field, "")).strip().lower() == "true" for row in non_nano)
555
+ if excluded_ci_count != int(expected_excluded_ci):
556
+ errors.append(f"{summary_id}: retained non-Nano CI count {excluded_ci_count} != {expected_excluded_ci}")
557
+ else:
558
+ checked += 1
559
+ elif summary.get("kind") == "t2_holdout_summary":
560
+ expected_cells = int(summary.get("cells", -1))
561
+ if len(rows) != expected_cells:
562
+ errors.append(f"{summary_id}: retained T2 row count {len(rows)} != {expected_cells}")
563
+ continue
564
+ checked += 1
565
+ def _count_true(field: str) -> int:
566
+ return sum(str(row.get(field, "")).strip().lower() == "true" for row in rows)
567
+ checks = (("compatible_cells", "compatible_eps_001"), ("safe_predictions", "prediction"), ("observed_false_safe_predictions", "false_safe"))
568
+ for summary_key, field in checks:
569
+ if summary_key == "safe_predictions":
570
+ actual = sum(str(row.get(field, "")).strip().upper() == "SAFE" for row in rows)
571
+ else:
572
+ actual = _count_true(field)
573
+ expected_value = int(summary.get(summary_key, -1))
574
+ if actual != expected_value:
575
+ errors.append(f"{summary_id}: retained {summary_key} count {actual} != {expected_value}")
576
+ else:
577
+ checked += 1
578
+ # Verify the four published source/target group totals when
579
+ # the ledger carries those identifiers. This catches a row
580
+ # swap that preserves only the aggregate 28/22/17/0 counts.
581
+ aliases = {"qwen3_0_6b": "Qwen3-Embedding-0.6B", "qwen3_4b": "Qwen3-Embedding-4B", "qwen3_8b": "Qwen3-Embedding-8B", "minilm_l6": "MiniLM-L6"}
582
+ for group in summary.get("groups", []):
583
+ expected_source = str(group.get("source", ""))
584
+ expected_target = str(group.get("target", ""))
585
+ selected = [row for row in rows if aliases.get(str(row.get("source_model")), str(row.get("source_model"))) == expected_source and aliases.get(str(row.get("target_model")), str(row.get("target_model"))) == expected_target]
586
+ expected_group_cells = int(group.get("cells", -1))
587
+ if len(selected) != expected_group_cells:
588
+ errors.append(f"{summary_id}: group {expected_source}->{expected_target} has {len(selected)} rows != {expected_group_cells}")
589
+ continue
590
+ for key, field, predicate in (("compatible", "compatible_eps_001", lambda value: str(value).lower() == "true"), ("safe", "prediction", lambda value: str(value).upper() == "SAFE"), ("false_safe", "false_safe", lambda value: str(value).lower() == "true")):
591
+ actual = sum(predicate(row.get(field, "")) for row in selected)
592
+ if actual != int(group.get(key, -1)):
593
+ errors.append(f"{summary_id}: group {expected_source}->{expected_target} {key} count {actual} != {group.get(key)}")
594
+ else:
595
+ checked += 1
596
+ return checked
597
+
598
+
599
+ def verify_registry(path: str | Path | None = None, *, provenance_root: str | Path | None = None) -> dict[str, Any]:
600
+ """Validate schema, hashes, consistency, and provenance metadata.
601
+
602
+ Provenance paths point to the retained research checkout and are not
603
+ expected to exist in an installed package. The recorded artifact digest
604
+ is checked for shape, while the packaged bytes are checked by the manifest.
605
+ When ``provenance_root`` is available, known canonical CSV summaries are
606
+ also checked field-by-field against the transcribed registry values.
607
+ """
608
+ errors: list[str] = []
609
+ warnings: list[str] = []
610
+ if path is None:
611
+ root = _resource_root()
612
+ manifest = load_manifest()
613
+ else:
614
+ root = _root_for(path)
615
+ try:
616
+ manifest = load_manifest(path)
617
+ except Exception as exc:
618
+ return {"ok": False, "errors": [str(exc)], "warnings": [], "record_count": 0, "profile_count": 0}
619
+ if str(manifest.get("schema_version")) != SCHEMA_VERSION:
620
+ errors.append(f"unsupported schema_version: {manifest.get('schema_version')!r}")
621
+ if str(manifest.get("registry_version")) != REGISTRY_VERSION:
622
+ errors.append(f"unsupported registry_version: {manifest.get('registry_version')!r}")
623
+ # Keep the standalone schema marker meaningful. It is shipped alongside
624
+ # the data so downstream tooling can reject a registry before attempting
625
+ # to parse rows.
626
+ try:
627
+ schema_marker = _read_json("schema_version.json", root)
628
+ if str(schema_marker.get("schema_version")) != SCHEMA_VERSION:
629
+ errors.append(f"schema_version.json declares {schema_marker.get('schema_version')!r}, expected {SCHEMA_VERSION!r}")
630
+ if str(schema_marker.get("registry_version")) != REGISTRY_VERSION:
631
+ errors.append(f"schema_version.json declares registry_version {schema_marker.get('registry_version')!r}, expected {REGISTRY_VERSION!r}")
632
+ except Exception as exc:
633
+ errors.append(str(exc))
634
+ checked_files = _verify_checksums(root, manifest, errors)
635
+ try:
636
+ evidence = load_evidence(path)
637
+ except Exception as exc:
638
+ evidence, profiles = [], []
639
+ errors.append(str(exc))
640
+ else:
641
+ try:
642
+ profiles = load_benchmark_profiles(path)
643
+ except Exception as exc:
644
+ profiles = []
645
+ errors.append(str(exc))
646
+ expected_count = manifest.get("record_count")
647
+ if expected_count is not None:
648
+ try:
649
+ count_value = int(expected_count)
650
+ except (TypeError, ValueError, OverflowError):
651
+ errors.append("manifest record_count must be an integer")
652
+ else:
653
+ if count_value != len(evidence):
654
+ errors.append(f"manifest record_count={expected_count} but loaded {len(evidence)}")
655
+ expected_profiles = manifest.get("profile_count")
656
+ if expected_profiles is not None:
657
+ try:
658
+ profiles_value = int(expected_profiles)
659
+ except (TypeError, ValueError, OverflowError):
660
+ errors.append("manifest profile_count must be an integer")
661
+ else:
662
+ if profiles_value != len(profiles):
663
+ errors.append(f"manifest profile_count={expected_profiles} but loaded {len(profiles)}")
664
+ try:
665
+ summaries = load_summaries(path)
666
+ except Exception as exc:
667
+ summaries = []
668
+ errors.append(str(exc))
669
+ expected_summaries = manifest.get("summary_count")
670
+ if expected_summaries is not None:
671
+ try:
672
+ summaries_value = int(expected_summaries)
673
+ except (TypeError, ValueError, OverflowError):
674
+ errors.append("manifest summary_count must be an integer")
675
+ else:
676
+ if summaries_value != len(summaries):
677
+ errors.append(f"manifest summary_count={expected_summaries} but loaded {len(summaries)}")
678
+ for summary in summaries:
679
+ if not isinstance(summary, dict) or not str(summary.get("summary_id", "")).strip():
680
+ errors.append("research summary lacks summary_id")
681
+ continue
682
+ provenance = summary.get("provenance") or {}
683
+ for digest_name in ("artifact_sha256", "supporting_artifact_sha256"):
684
+ digest = provenance.get(digest_name)
685
+ if digest is not None and (not isinstance(digest, str) or len(digest) != 64 or any(c not in "0123456789abcdef" for c in digest.lower())):
686
+ errors.append(f"{summary['summary_id']}: invalid provenance {digest_name} digest")
687
+ ids = [row.evidence_id for row in evidence]
688
+ if len(ids) != len(set(ids)):
689
+ errors.append("evidence_id values are not unique")
690
+ profile_ids = [profile.profile_id for profile in profiles]
691
+ if len(profile_ids) != len(set(profile_ids)):
692
+ errors.append("profile_id values are not unique")
693
+ for row in evidence:
694
+ provenance = row.raw.get("provenance", {})
695
+ digest = provenance.get("artifact_sha256")
696
+ if digest is not None and (not isinstance(digest, str) or len(digest) != 64 or any(c not in "0123456789abcdef" for c in digest.lower())):
697
+ errors.append(f"{row.evidence_id}: provenance.artifact_sha256 is not a SHA-256 digest")
698
+ if provenance.get("status") not in {"retained_external", "packaged", "derived_from_retained_external"}:
699
+ errors.append(f"{row.evidence_id}: provenance.status must declare whether the artifact is retained externally")
700
+ ann = row.raw.get("ann") or {}
701
+ if ann.get("status") == "UNKNOWN" and ann.get("tested_configurations"):
702
+ warnings.append(f"{row.evidence_id}: ANN status UNKNOWN despite tested configurations; review semantics")
703
+ for profile in profiles:
704
+ provenance = profile.raw.get("provenance", {})
705
+ digest = provenance.get("artifact_sha256")
706
+ if digest is not None and (not isinstance(digest, str) or len(digest) != 64):
707
+ errors.append(f"{profile.profile_id}: invalid provenance artifact digest")
708
+ provenance_checked = _verify_external_provenance(provenance_root, evidence, profiles, summaries, errors, warnings)
709
+ semantic_checked = _verify_semantic_provenance(provenance_root, evidence, profiles, summaries, errors, warnings)
710
+ return {
711
+ "ok": not errors,
712
+ "errors": errors,
713
+ "warnings": warnings,
714
+ "record_count": len(evidence),
715
+ "profile_count": len(profiles),
716
+ "summary_count": len(summaries),
717
+ "checked_files": checked_files,
718
+ "provenance_artifacts_checked": provenance_checked,
719
+ "semantic_values_checked": semantic_checked,
720
+ "registry_version": manifest.get("registry_version"),
721
+ "schema_version": manifest.get("schema_version"),
722
+ }
723
+
724
+
725
+ def _canonical_record(value: Mapping[str, Any]) -> bytes:
726
+ return (json.dumps(dict(value), sort_keys=True, separators=(",", ":"), ensure_ascii=False) + "\n").encode()
727
+
728
+
729
+ def dataset_fingerprint(
730
+ path: str | Path,
731
+ *,
732
+ id_field: str = "id",
733
+ text_field: str = "text",
734
+ query_path: str | Path | None = None,
735
+ qrels_path: str | Path | None = None,
736
+ ) -> str:
737
+ """Create a stable fingerprint for JSONL corpus/query/qrels construction.
738
+
739
+ The hash includes field names, canonicalized row values, and row counts.
740
+ It is deliberately not inferred from a dataset name alone.
741
+ """
742
+ digest = hashlib.sha256()
743
+ count = 0
744
+ digest.update(f"documents:{id_field}:{text_field}\n".encode())
745
+ with Path(path).open(encoding="utf-8") as handle:
746
+ for line_number, line in enumerate(handle, 1):
747
+ if not line.strip():
748
+ continue
749
+ try:
750
+ row = json.loads(line)
751
+ except json.JSONDecodeError as exc:
752
+ raise ValueError(f"invalid JSONL at {path}:{line_number}") from exc
753
+ if not isinstance(row, dict) or id_field not in row or text_field not in row:
754
+ raise ValueError(f"row {line_number} in {path} lacks {id_field!r}/{text_field!r}")
755
+ digest.update(_canonical_record({"id": str(row[id_field]), "text": str(row[text_field])}))
756
+ count += 1
757
+ digest.update(f"documents_count:{count}\n".encode())
758
+ for label, extra in (("queries", query_path), ("qrels", qrels_path)):
759
+ if extra is None:
760
+ continue
761
+ extra_digest = hashlib.sha256(Path(extra).read_bytes()).hexdigest()
762
+ digest.update(f"{label}:{extra_digest}\n".encode())
763
+ return digest.hexdigest()
764
+
765
+
766
+ def find_evidence(
767
+ *,
768
+ source_model: Any,
769
+ target_model: Any,
770
+ corpus_fingerprint: str | None = None,
771
+ corpus_name: str | None = None,
772
+ corpus_size: int | None = None,
773
+ records: Iterable[EvidenceRecord] | None = None,
774
+ ):
775
+ """Find evidence using the public matching API (lazy import avoids cycles)."""
776
+ from .matcher import match_evidence
777
+
778
+ return match_evidence(
779
+ source_model=source_model,
780
+ target_model=target_model,
781
+ corpus_fingerprint=corpus_fingerprint,
782
+ corpus_name=corpus_name,
783
+ corpus_size=corpus_size,
784
+ records=list(records) if records is not None else load_evidence(),
785
+ )