reader-workbench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (293) hide show
  1. reader_workbench/__init__.py +22 -0
  2. reader_workbench/__main__.py +4 -0
  3. reader_workbench/_version.py +17 -0
  4. reader_workbench/api/__init__.py +74 -0
  5. reader_workbench/api/_record_reads.py +75 -0
  6. reader_workbench/api/artifacts.py +79 -0
  7. reader_workbench/api/facade.py +538 -0
  8. reader_workbench/api/models.py +285 -0
  9. reader_workbench/api/notebooks.py +63 -0
  10. reader_workbench/contracts/__init__.py +18 -0
  11. reader_workbench/contracts/builtins/__init__.py +36 -0
  12. reader_workbench/contracts/builtins/cytometry.py +140 -0
  13. reader_workbench/contracts/builtins/four_state_event_window.py +214 -0
  14. reader_workbench/contracts/builtins/generic.py +21 -0
  15. reader_workbench/contracts/builtins/logic.py +149 -0
  16. reader_workbench/contracts/builtins/plate_reader.py +47 -0
  17. reader_workbench/contracts/catalog.py +257 -0
  18. reader_workbench/contracts/model.py +109 -0
  19. reader_workbench/domains/__init__.py +1 -0
  20. reader_workbench/domains/cytometry/__init__.py +3 -0
  21. reader_workbench/domains/cytometry/analysis/__init__.py +29 -0
  22. reader_workbench/domains/cytometry/analysis/events.py +182 -0
  23. reader_workbench/domains/cytometry/analysis/gating.py +175 -0
  24. reader_workbench/domains/cytometry/analysis/workflow.py +248 -0
  25. reader_workbench/domains/cytometry/io/__init__.py +3 -0
  26. reader_workbench/domains/cytometry/io/fcs.py +135 -0
  27. reader_workbench/domains/cytometry/plots/__init__.py +5 -0
  28. reader_workbench/domains/cytometry/plots/diagnostic.py +155 -0
  29. reader_workbench/domains/logic/__init__.py +3 -0
  30. reader_workbench/domains/logic/crosstalk/__init__.py +3 -0
  31. reader_workbench/domains/logic/crosstalk/pairs.py +661 -0
  32. reader_workbench/domains/logic/four_state_vector/__init__.py +7 -0
  33. reader_workbench/domains/logic/four_state_vector/builder.py +321 -0
  34. reader_workbench/domains/logic/four_state_vector/collection/__init__.py +20 -0
  35. reader_workbench/domains/logic/four_state_vector/collection/checks.py +49 -0
  36. reader_workbench/domains/logic/four_state_vector/collection/constants.py +29 -0
  37. reader_workbench/domains/logic/four_state_vector/collection/model.py +19 -0
  38. reader_workbench/domains/logic/four_state_vector/collection/render.py +383 -0
  39. reader_workbench/domains/logic/four_state_vector/collection/sources.py +185 -0
  40. reader_workbench/domains/logic/four_state_vector/config.py +214 -0
  41. reader_workbench/domains/logic/four_state_vector/diagnostic.py +361 -0
  42. reader_workbench/domains/logic/four_state_vector/heatmap.py +86 -0
  43. reader_workbench/domains/logic/four_state_vector/math.py +191 -0
  44. reader_workbench/domains/logic/four_state_vector/reference.py +85 -0
  45. reader_workbench/domains/logic/four_state_vector/selection.py +228 -0
  46. reader_workbench/domains/logic/four_state_vector/treatment_semantics.py +51 -0
  47. reader_workbench/domains/logic/four_state_vector/validation.py +19 -0
  48. reader_workbench/domains/logic/logic_symmetry/__init__.py +3 -0
  49. reader_workbench/domains/logic/logic_symmetry/encodings.py +93 -0
  50. reader_workbench/domains/logic/logic_symmetry/extract_corners.py +192 -0
  51. reader_workbench/domains/logic/logic_symmetry/main.py +236 -0
  52. reader_workbench/domains/logic/logic_symmetry/metrics.py +96 -0
  53. reader_workbench/domains/logic/logic_symmetry/overlay.py +129 -0
  54. reader_workbench/domains/logic/logic_symmetry/prep.py +138 -0
  55. reader_workbench/domains/logic/logic_symmetry/render.py +356 -0
  56. reader_workbench/domains/logic/treatment_columns.py +42 -0
  57. reader_workbench/domains/plate_reader/__init__.py +1 -0
  58. reader_workbench/domains/plate_reader/analysis/__init__.py +14 -0
  59. reader_workbench/domains/plate_reader/analysis/fold_change.py +474 -0
  60. reader_workbench/domains/plate_reader/analysis/four_state_event_window/__init__.py +21 -0
  61. reader_workbench/domains/plate_reader/analysis/four_state_event_window/aggregation.py +191 -0
  62. reader_workbench/domains/plate_reader/analysis/four_state_event_window/contract_fields.py +50 -0
  63. reader_workbench/domains/plate_reader/analysis/four_state_event_window/contracts.py +320 -0
  64. reader_workbench/domains/plate_reader/analysis/four_state_event_window/design_dispositions.py +54 -0
  65. reader_workbench/domains/plate_reader/analysis/four_state_event_window/disposition_records.py +140 -0
  66. reader_workbench/domains/plate_reader/analysis/four_state_event_window/event_sensitivity.py +27 -0
  67. reader_workbench/domains/plate_reader/analysis/four_state_event_window/materialize.py +274 -0
  68. reader_workbench/domains/plate_reader/analysis/four_state_event_window/observation_resampling.py +97 -0
  69. reader_workbench/domains/plate_reader/analysis/four_state_event_window/reduction.py +62 -0
  70. reader_workbench/domains/plate_reader/analysis/four_state_event_window/seeds.py +15 -0
  71. reader_workbench/domains/plate_reader/analysis/four_state_event_window/sources.py +308 -0
  72. reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusion_validation.py +94 -0
  73. reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusions.py +54 -0
  74. reader_workbench/domains/plate_reader/analysis/timepoints.py +76 -0
  75. reader_workbench/domains/plate_reader/io/__init__.py +6 -0
  76. reader_workbench/domains/plate_reader/io/sample_map.py +65 -0
  77. reader_workbench/domains/plate_reader/io/synergy_h1/__init__.py +6 -0
  78. reader_workbench/domains/plate_reader/io/synergy_h1/_kinetic.py +141 -0
  79. reader_workbench/domains/plate_reader/io/synergy_h1/_parser.py +295 -0
  80. reader_workbench/domains/plate_reader/io/synergy_h1/_shared.py +195 -0
  81. reader_workbench/domains/plate_reader/io/synergy_h1/_snapshot.py +130 -0
  82. reader_workbench/domains/plate_reader/ordering.py +59 -0
  83. reader_workbench/domains/plate_reader/plots/__init__.py +15 -0
  84. reader_workbench/domains/plate_reader/plots/_data.py +29 -0
  85. reader_workbench/domains/plate_reader/plots/common.py +346 -0
  86. reader_workbench/domains/plate_reader/plots/distributions.py +324 -0
  87. reader_workbench/domains/plate_reader/plots/dual_reporter_triptych.py +525 -0
  88. reader_workbench/domains/plate_reader/plots/dual_reporter_triptych_render.py +195 -0
  89. reader_workbench/domains/plate_reader/plots/four_state_event_window/__init__.py +26 -0
  90. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic.py +295 -0
  91. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_components.py +149 -0
  92. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_render.py +308 -0
  93. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_style.py +51 -0
  94. reader_workbench/domains/plate_reader/plots/four_state_event_window/schema.py +8 -0
  95. reader_workbench/domains/plate_reader/plots/four_state_event_window/summary.py +140 -0
  96. reader_workbench/domains/plate_reader/plots/grouping.py +53 -0
  97. reader_workbench/domains/plate_reader/plots/panels/__init__.py +12 -0
  98. reader_workbench/domains/plate_reader/plots/panels/snapshot.py +161 -0
  99. reader_workbench/domains/plate_reader/plots/panels/snapshot_data.py +91 -0
  100. reader_workbench/domains/plate_reader/plots/panels/time_series.py +296 -0
  101. reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic.py +435 -0
  102. reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic_render.py +300 -0
  103. reader_workbench/domains/plate_reader/plots/snapshot_barplot/__init__.py +315 -0
  104. reader_workbench/domains/plate_reader/plots/snapshot_barplot/planning.py +168 -0
  105. reader_workbench/domains/plate_reader/plots/snapshot_heatmap/__init__.py +205 -0
  106. reader_workbench/domains/plate_reader/plots/snapshot_heatmap/inputs.py +103 -0
  107. reader_workbench/domains/plate_reader/plots/time_series.py +317 -0
  108. reader_workbench/domains/plate_reader/plots/ts_and_snap/__init__.py +479 -0
  109. reader_workbench/domains/plate_reader/plots/ts_and_snap/planning.py +283 -0
  110. reader_workbench/domains/time_series/__init__.py +29 -0
  111. reader_workbench/domains/time_series/aggregation.py +60 -0
  112. reader_workbench/domains/time_series/contracts.py +368 -0
  113. reader_workbench/domains/time_series/reduction.py +395 -0
  114. reader_workbench/errors.py +55 -0
  115. reader_workbench/maintenance/__init__.py +6 -0
  116. reader_workbench/maintenance/docs.py +335 -0
  117. reader_workbench/maintenance/model.py +28 -0
  118. reader_workbench/maintenance/release.py +39 -0
  119. reader_workbench/maintenance/skills.py +124 -0
  120. reader_workbench/plotting/__init__.py +20 -0
  121. reader_workbench/plotting/mpl.py +56 -0
  122. reader_workbench/plotting/sinks.py +69 -0
  123. reader_workbench/plotting/style.py +175 -0
  124. reader_workbench/plotting/utils.py +27 -0
  125. reader_workbench/plugins/__init__.py +1 -0
  126. reader_workbench/plugins/catalog.py +33 -0
  127. reader_workbench/plugins/export/__init__.py +0 -0
  128. reader_workbench/plugins/export/_paths.py +21 -0
  129. reader_workbench/plugins/export/csv.py +41 -0
  130. reader_workbench/plugins/export/xlsx.py +44 -0
  131. reader_workbench/plugins/ingest/__init__.py +0 -0
  132. reader_workbench/plugins/ingest/_discovery.py +58 -0
  133. reader_workbench/plugins/ingest/discovery_policy.py +66 -0
  134. reader_workbench/plugins/ingest/flow_cytometer.py +139 -0
  135. reader_workbench/plugins/ingest/synergy_h1.py +234 -0
  136. reader_workbench/plugins/manifests/__init__.py +1 -0
  137. reader_workbench/plugins/manifests/export.py +29 -0
  138. reader_workbench/plugins/manifests/ingest.py +29 -0
  139. reader_workbench/plugins/manifests/plot.py +161 -0
  140. reader_workbench/plugins/manifests/transform.py +172 -0
  141. reader_workbench/plugins/manifests/validator.py +18 -0
  142. reader_workbench/plugins/plot/__init__.py +0 -0
  143. reader_workbench/plugins/plot/_shared.py +55 -0
  144. reader_workbench/plugins/plot/cytometry_diagnostic.py +54 -0
  145. reader_workbench/plugins/plot/distributions.py +58 -0
  146. reader_workbench/plugins/plot/dual_reporter_triptych.py +224 -0
  147. reader_workbench/plugins/plot/four_state_event_window_diagnostic.py +133 -0
  148. reader_workbench/plugins/plot/four_state_event_window_summary.py +53 -0
  149. reader_workbench/plugins/plot/four_state_vector_collection.py +45 -0
  150. reader_workbench/plugins/plot/four_state_vector_diagnostic.py +105 -0
  151. reader_workbench/plugins/plot/four_state_vector_heatmap.py +63 -0
  152. reader_workbench/plugins/plot/logic_symmetry.py +56 -0
  153. reader_workbench/plugins/plot/single_reporter_diagnostic.py +279 -0
  154. reader_workbench/plugins/plot/snapshot_barplot.py +65 -0
  155. reader_workbench/plugins/plot/snapshot_heatmap.py +104 -0
  156. reader_workbench/plugins/plot/time_series.py +114 -0
  157. reader_workbench/plugins/plot/ts_and_snap.py +210 -0
  158. reader_workbench/plugins/transform/__init__.py +0 -0
  159. reader_workbench/plugins/transform/_four_state_vector.py +204 -0
  160. reader_workbench/plugins/transform/_labeling.py +109 -0
  161. reader_workbench/plugins/transform/alias.py +70 -0
  162. reader_workbench/plugins/transform/assay_labels.py +62 -0
  163. reader_workbench/plugins/transform/blank.py +79 -0
  164. reader_workbench/plugins/transform/crosstalk_pairs.py +180 -0
  165. reader_workbench/plugins/transform/cytometry_gating.py +120 -0
  166. reader_workbench/plugins/transform/fold_change.py +79 -0
  167. reader_workbench/plugins/transform/four_state_event_window.py +93 -0
  168. reader_workbench/plugins/transform/four_state_vector.py +62 -0
  169. reader_workbench/plugins/transform/four_state_vector_collection.py +41 -0
  170. reader_workbench/plugins/transform/logic_symmetry.py +67 -0
  171. reader_workbench/plugins/transform/outlier_filter.py +60 -0
  172. reader_workbench/plugins/transform/overflow.py +197 -0
  173. reader_workbench/plugins/transform/ratio.py +237 -0
  174. reader_workbench/plugins/transform/sample_map.py +170 -0
  175. reader_workbench/plugins/transform/sample_metadata.py +94 -0
  176. reader_workbench/plugins/validator/__init__.py +1 -0
  177. reader_workbench/plugins/validator/to_tidy_plus_map.py +155 -0
  178. reader_workbench/protocols/__init__.py +80 -0
  179. reader_workbench/protocols/_builtins_plate_reader_growth.py +179 -0
  180. reader_workbench/protocols/_builtins_plate_reader_variants.py +274 -0
  181. reader_workbench/protocols/builtins.py +1656 -0
  182. reader_workbench/protocols/compiler.py +22 -0
  183. reader_workbench/protocols/compilers/__init__.py +1 -0
  184. reader_workbench/protocols/compilers/common.py +100 -0
  185. reader_workbench/protocols/compilers/cytometry.py +87 -0
  186. reader_workbench/protocols/compilers/generic.py +14 -0
  187. reader_workbench/protocols/compilers/logic.py +245 -0
  188. reader_workbench/protocols/compilers/plate_reader.py +937 -0
  189. reader_workbench/protocols/compilers/plate_reader_pipeline.py +197 -0
  190. reader_workbench/protocols/model.py +1486 -0
  191. reader_workbench/protocols/semantic_coverage.py +234 -0
  192. reader_workbench/runtime/__init__.py +12 -0
  193. reader_workbench/runtime/builtin.py +23 -0
  194. reader_workbench/runtime/model.py +42 -0
  195. reader_workbench/workbench/__init__.py +60 -0
  196. reader_workbench/workbench/assets/__init__.py +22 -0
  197. reader_workbench/workbench/assets/types.py +118 -0
  198. reader_workbench/workbench/audit/__init__.py +5 -0
  199. reader_workbench/workbench/audit/experiments.py +307 -0
  200. reader_workbench/workbench/audit/staging.py +187 -0
  201. reader_workbench/workbench/cli/__init__.py +51 -0
  202. reader_workbench/workbench/cli/_lazy.py +9 -0
  203. reader_workbench/workbench/cli/_records_view.py +150 -0
  204. reader_workbench/workbench/cli/_surface_execution.py +443 -0
  205. reader_workbench/workbench/cli/audit.py +95 -0
  206. reader_workbench/workbench/cli/automation.py +229 -0
  207. reader_workbench/workbench/cli/demo.py +46 -0
  208. reader_workbench/workbench/cli/dop.py +91 -0
  209. reader_workbench/workbench/cli/experiments.py +635 -0
  210. reader_workbench/workbench/cli/helpers.py +232 -0
  211. reader_workbench/workbench/cli/main.py +59 -0
  212. reader_workbench/workbench/cli/maintenance.py +82 -0
  213. reader_workbench/workbench/cli/notebooks.py +260 -0
  214. reader_workbench/workbench/cli/pagination.py +117 -0
  215. reader_workbench/workbench/cli/protocols.py +336 -0
  216. reader_workbench/workbench/cli/shared.py +309 -0
  217. reader_workbench/workbench/cli/surfaces.py +534 -0
  218. reader_workbench/workbench/cli/verification.py +128 -0
  219. reader_workbench/workbench/commands.py +10 -0
  220. reader_workbench/workbench/config/__init__.py +47 -0
  221. reader_workbench/workbench/config/identity.py +13 -0
  222. reader_workbench/workbench/config/load.py +405 -0
  223. reader_workbench/workbench/config/model.py +274 -0
  224. reader_workbench/workbench/context.py +26 -0
  225. reader_workbench/workbench/decl/__init__.py +31 -0
  226. reader_workbench/workbench/decl/build.py +190 -0
  227. reader_workbench/workbench/decl/model.py +81 -0
  228. reader_workbench/workbench/dop/__init__.py +12 -0
  229. reader_workbench/workbench/dop/builtins.py +261 -0
  230. reader_workbench/workbench/dop/model.py +209 -0
  231. reader_workbench/workbench/engine/__init__.py +42 -0
  232. reader_workbench/workbench/engine/_shared.py +76 -0
  233. reader_workbench/workbench/engine/contracts.py +283 -0
  234. reader_workbench/workbench/engine/execution.py +326 -0
  235. reader_workbench/workbench/engine/file_outputs.py +260 -0
  236. reader_workbench/workbench/engine/inputs.py +161 -0
  237. reader_workbench/workbench/engine/invocations.py +507 -0
  238. reader_workbench/workbench/engine/planning.py +72 -0
  239. reader_workbench/workbench/engine/runtime.py +464 -0
  240. reader_workbench/workbench/engine/setup.py +149 -0
  241. reader_workbench/workbench/engine/validation.py +684 -0
  242. reader_workbench/workbench/experiment/__init__.py +47 -0
  243. reader_workbench/workbench/experiment/model.py +381 -0
  244. reader_workbench/workbench/experiments.py +133 -0
  245. reader_workbench/workbench/graph/__init__.py +47 -0
  246. reader_workbench/workbench/graph/nodes.py +102 -0
  247. reader_workbench/workbench/graph/normalize.py +177 -0
  248. reader_workbench/workbench/graph/refs.py +148 -0
  249. reader_workbench/workbench/input_discovery.py +19 -0
  250. reader_workbench/workbench/inspection/__init__.py +3 -0
  251. reader_workbench/workbench/inspection/catalogs.py +128 -0
  252. reader_workbench/workbench/inspection/common.py +92 -0
  253. reader_workbench/workbench/inspection/dop.py +64 -0
  254. reader_workbench/workbench/inspection/experiments.py +449 -0
  255. reader_workbench/workbench/inspection/inventory.py +68 -0
  256. reader_workbench/workbench/inspection/protocols.py +368 -0
  257. reader_workbench/workbench/inspection/readiness.py +333 -0
  258. reader_workbench/workbench/inspection/reports.py +367 -0
  259. reader_workbench/workbench/inspection/results.py +166 -0
  260. reader_workbench/workbench/inspection/runtime.py +287 -0
  261. reader_workbench/workbench/inspection/semantics.py +192 -0
  262. reader_workbench/workbench/inspection/validation.py +30 -0
  263. reader_workbench/workbench/notebooks/__init__.py +17 -0
  264. reader_workbench/workbench/notebooks/_launch_registry.py +112 -0
  265. reader_workbench/workbench/notebooks/_launch_runtime.py +104 -0
  266. reader_workbench/workbench/notebooks/components/__init__.py +21 -0
  267. reader_workbench/workbench/notebooks/components/deliverables.py +403 -0
  268. reader_workbench/workbench/notebooks/components/overview.py +119 -0
  269. reader_workbench/workbench/notebooks/eda.marimo.py.txt +153 -0
  270. reader_workbench/workbench/notebooks/launch.py +274 -0
  271. reader_workbench/workbench/notebooks/presentation.py +136 -0
  272. reader_workbench/workbench/notebooks/scaffold.py +60 -0
  273. reader_workbench/workbench/ontology.py +78 -0
  274. reader_workbench/workbench/paths.py +44 -0
  275. reader_workbench/workbench/ports/__init__.py +31 -0
  276. reader_workbench/workbench/ports/model.py +168 -0
  277. reader_workbench/workbench/records/__init__.py +44 -0
  278. reader_workbench/workbench/records/epoch.py +329 -0
  279. reader_workbench/workbench/records/evidence.py +247 -0
  280. reader_workbench/workbench/records/identity.py +87 -0
  281. reader_workbench/workbench/records/locking.py +185 -0
  282. reader_workbench/workbench/records/model.py +711 -0
  283. reader_workbench/workbench/records/sources.py +73 -0
  284. reader_workbench/workbench/records/store.py +1022 -0
  285. reader_workbench/workbench/records/verification.py +998 -0
  286. reader_workbench/workbench/registry.py +333 -0
  287. reader_workbench/workbench/spec_overrides.py +215 -0
  288. reader_workbench-1.0.0.dist-info/METADATA +91 -0
  289. reader_workbench-1.0.0.dist-info/RECORD +293 -0
  290. reader_workbench-1.0.0.dist-info/WHEEL +5 -0
  291. reader_workbench-1.0.0.dist-info/entry_points.txt +2 -0
  292. reader_workbench-1.0.0.dist-info/licenses/LICENSE +21 -0
  293. reader_workbench-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1022 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import os
5
+ import stat
6
+ from collections.abc import Iterable, Iterator, Mapping
7
+ from contextlib import contextmanager, suppress
8
+ from dataclasses import dataclass
9
+ from datetime import UTC, datetime
10
+ from pathlib import Path
11
+ from tempfile import NamedTemporaryFile
12
+ from typing import Any
13
+ from uuid import UUID, uuid4
14
+
15
+ import pandas as pd
16
+
17
+ from reader_workbench.contracts import ContractCatalog, ContractId
18
+ from reader_workbench.errors import ProvenanceEpochChangedError, RecordError
19
+ from reader_workbench.workbench.experiments import ExperimentCatalog, ExperimentLocation
20
+ from reader_workbench.workbench.graph import (
21
+ FileRef,
22
+ ProvenanceInput,
23
+ RecipeSource,
24
+ RecordRef,
25
+ ResourceRef,
26
+ SourceRecordRef,
27
+ )
28
+ from reader_workbench.workbench.ontology import WorkbenchProducerKind, WorkbenchRecordKind
29
+ from reader_workbench.workbench.paths import resolve_confined_sink_root
30
+ from reader_workbench.workbench.records.model import (
31
+ DataFrameArtifactRecord,
32
+ FileBundleRecord,
33
+ PathDescription,
34
+ RecordProducer,
35
+ RecordRecipeSource,
36
+ normalize_file_bundle_metadata,
37
+ record_from_dict,
38
+ record_to_dict,
39
+ sha256_file,
40
+ verify_record_artifact_integrity,
41
+ )
42
+
43
+ from .epoch import assert_no_interrupted_epoch, replace_generated_epoch, validate_generated_epoch_boundary
44
+ from .evidence import RecordInputEvidence, SourceExperimentResolver, capture_artifact_evidence
45
+ from .identity import BuildIdentity, current_build_identity, digest_json
46
+ from .locking import ProvenanceFileLock, provenance_lock_scope
47
+ from .model import record_revision_digest
48
+
49
+ RECORD_CATALOG_SCHEMA_VERSION = 4
50
+ _RECORD_CATALOG_FIELDS = frozenset({"schema_version", "provenance_epoch_id", "latest", "history"})
51
+
52
+
53
+ def _is_canonical_uuid4(value: object) -> bool:
54
+ if not isinstance(value, str):
55
+ return False
56
+ try:
57
+ parsed = UUID(value)
58
+ except ValueError:
59
+ return False
60
+ return parsed.version == 4 and str(parsed) == value
61
+
62
+
63
+ def _empty_catalog() -> dict[str, Any]:
64
+ return {
65
+ "schema_version": RECORD_CATALOG_SCHEMA_VERSION,
66
+ "provenance_epoch_id": str(uuid4()),
67
+ "latest": {},
68
+ "history": {},
69
+ }
70
+
71
+
72
+ def _selected_latest_record_ids(
73
+ latest: Mapping[str, Any],
74
+ *,
75
+ config_digest: str | None,
76
+ record_ids: frozenset[str] | None,
77
+ ) -> frozenset[str]:
78
+ candidates = set(latest)
79
+ if record_ids is not None:
80
+ candidates &= record_ids
81
+ if config_digest is None:
82
+ return frozenset(candidates)
83
+ return frozenset(
84
+ record_id
85
+ for record_id in candidates
86
+ if not isinstance(latest[record_id], dict) or latest[record_id].get("config_digest") == config_digest
87
+ )
88
+
89
+
90
+ @dataclass(frozen=True)
91
+ class RecordCatalogSnapshot:
92
+ schema_version: int
93
+ provenance_epoch_id: str
94
+ active_invocation_ledger: Path
95
+ latest_records: tuple[DataFrameArtifactRecord | FileBundleRecord, ...]
96
+ revision_counts: dict[str, int]
97
+
98
+
99
+ class RecordStore:
100
+ def __init__(
101
+ self,
102
+ outputs_dir: Path,
103
+ *,
104
+ contracts: ContractCatalog,
105
+ plots_subdir: str | None = "plots",
106
+ exports_subdir: str | None = "exports",
107
+ experiment_root: Path | None = None,
108
+ create: bool = True,
109
+ ) -> None:
110
+ self.root = Path(outputs_dir).expanduser().absolute()
111
+ self.contracts = contracts
112
+ self.artifacts_dir = self.root / "artifacts"
113
+ self.manifests_dir = self.root / "manifests"
114
+ self.records_path = self.manifests_dir / "records.json"
115
+ self._catalog_lock_path = self.manifests_dir / ".records.lock"
116
+ self._catalog_lock = ProvenanceFileLock(self._catalog_lock_path, timeout=30)
117
+ self._bound_provenance_epoch_id: str | None = None
118
+ self.experiment_root = Path(experiment_root or self.root.parent).expanduser().absolute()
119
+ if plots_subdir in (None, "", ".", "./"):
120
+ self.plots_dir = self.root
121
+ else:
122
+ self.plots_dir = self.root / plots_subdir
123
+ if exports_subdir in (None, "", ".", "./"):
124
+ self.exports_dir = self.root
125
+ else:
126
+ self.exports_dir = self.root / exports_subdir
127
+ self._validate_sink_roots()
128
+ if create:
129
+ self.ensure_layout()
130
+
131
+ def _validate_sink_roots(self) -> None:
132
+ try:
133
+ resolve_confined_sink_root(self.root, root=self.experiment_root, label="outputs")
134
+ for label, path in (
135
+ ("artifacts", self.artifacts_dir),
136
+ ("manifests", self.manifests_dir),
137
+ ("plots", self.plots_dir),
138
+ ("exports", self.exports_dir),
139
+ ):
140
+ resolve_confined_sink_root(path, root=self.root, label=label)
141
+ except ValueError as exc:
142
+ raise RecordError(str(exc)) from exc
143
+
144
+ def _validate_catalog_roots(self) -> None:
145
+ try:
146
+ resolve_confined_sink_root(self.root, root=self.experiment_root, label="outputs")
147
+ resolve_confined_sink_root(self.manifests_dir, root=self.root, label="manifests")
148
+ except ValueError as exc:
149
+ raise RecordError(str(exc)) from exc
150
+
151
+ def ensure_layout(self) -> None:
152
+ self._validate_sink_roots()
153
+ self.artifacts_dir.mkdir(parents=True, exist_ok=True)
154
+ self.manifests_dir.mkdir(parents=True, exist_ok=True)
155
+ if self.plots_dir != self.root:
156
+ self.plots_dir.mkdir(parents=True, exist_ok=True)
157
+ if self.exports_dir != self.root:
158
+ self.exports_dir.mkdir(parents=True, exist_ok=True)
159
+ if not self.records_path.exists():
160
+ self._write_catalog(_empty_catalog(), create_only=True)
161
+
162
+ def catalog_exists(self) -> bool:
163
+ return self.records_path.exists() or self.records_path.is_symlink()
164
+
165
+ def provenance_lock_exists(self) -> bool:
166
+ """Return whether the non-followed writer coordination entry already exists."""
167
+
168
+ return self._catalog_lock_path.exists() or self._catalog_lock_path.is_symlink()
169
+
170
+ def reset_generated_epoch(self, *, preserved_paths: Iterable[Path] = ()) -> str:
171
+ """Begin a fresh epoch for catalog, artifacts, plots, and exports."""
172
+
173
+ if self._bound_provenance_epoch_id is not None:
174
+ raise RecordError("A RecordStore bound to a provenance epoch cannot reset generated outputs")
175
+
176
+ def _initialize() -> str:
177
+ self.ensure_layout()
178
+ return self.provenance_epoch_id()
179
+
180
+ with self._catalog_lock_scope(operation="reset the generated-output epoch"):
181
+ return replace_generated_epoch(
182
+ outputs_root=self.root,
183
+ artifacts_root=self.artifacts_dir,
184
+ manifests_root=self.manifests_dir,
185
+ plots_root=self.plots_dir,
186
+ exports_root=self.exports_dir,
187
+ log_path=self.root / "reader.log",
188
+ preserved_paths=preserved_paths,
189
+ lock_path=self._catalog_lock_path,
190
+ initialize=_initialize,
191
+ )
192
+
193
+ def validate_generated_epoch_reset(self, *, preserved_paths: Iterable[Path] = ()) -> None:
194
+ """Fail before mutation when generated-output ownership is unsafe to reset."""
195
+
196
+ validate_generated_epoch_boundary(
197
+ outputs_root=self.root,
198
+ artifacts_root=self.artifacts_dir,
199
+ manifests_root=self.manifests_dir,
200
+ plots_root=self.plots_dir,
201
+ exports_root=self.exports_dir,
202
+ log_path=self.root / "reader.log",
203
+ preserved_paths=preserved_paths,
204
+ lock_path=self._catalog_lock_path,
205
+ )
206
+
207
+ def validate_no_interrupted_epoch(self) -> None:
208
+ """Fail while recovery evidence from an interrupted reset remains."""
209
+
210
+ assert_no_interrupted_epoch(self.root)
211
+
212
+ def provenance_epoch_id(self) -> str:
213
+ """Return the catalog-owned identity for the active provenance epoch."""
214
+
215
+ return str(self._read_catalog()["provenance_epoch_id"])
216
+
217
+ def invocation_ledger_path(self) -> Path:
218
+ """Return the active epoch's deterministic invocation-ledger path."""
219
+
220
+ epoch_id = self.provenance_epoch_id()
221
+ return self.manifests_dir / "invocations" / f"{epoch_id}.jsonl"
222
+
223
+ def catalog_snapshot(
224
+ self,
225
+ *,
226
+ current_config_digest: str | None = None,
227
+ current_record_ids: frozenset[str] | None = None,
228
+ ) -> RecordCatalogSnapshot:
229
+ """Read one internally consistent catalog projection for inspection surfaces.
230
+
231
+ A current-config projection validates the full catalog structure and
232
+ every serialized payload, but resolves live external source locations
233
+ only for selected latest records. Historical and unselected source
234
+ resolution remains the responsibility of an unscoped snapshot.
235
+ """
236
+
237
+ source_experiment_resolver = self._source_experiment_resolver()
238
+ with self._catalog_lock_scope(operation="inspect the record catalog"):
239
+ catalog = self._read_catalog(
240
+ source_experiment_resolver=source_experiment_resolver,
241
+ materialize_config_digest=current_config_digest,
242
+ materialize_record_ids=current_record_ids,
243
+ )
244
+ epoch_id = str(catalog["provenance_epoch_id"])
245
+ selected_record_ids = _selected_latest_record_ids(
246
+ catalog["latest"],
247
+ config_digest=current_config_digest,
248
+ record_ids=current_record_ids,
249
+ )
250
+ records = tuple(
251
+ self._materialize(
252
+ record_id,
253
+ payload,
254
+ source_experiment_resolver=source_experiment_resolver,
255
+ )
256
+ for record_id, payload in sorted(catalog["latest"].items())
257
+ if record_id in selected_record_ids
258
+ )
259
+ revision_counts = {
260
+ record_id: len(catalog["history"][record_id]) for record_id in sorted(selected_record_ids)
261
+ }
262
+ return RecordCatalogSnapshot(
263
+ schema_version=RECORD_CATALOG_SCHEMA_VERSION,
264
+ provenance_epoch_id=epoch_id,
265
+ active_invocation_ledger=self.manifests_dir / "invocations" / f"{epoch_id}.jsonl",
266
+ latest_records=records,
267
+ revision_counts=revision_counts,
268
+ )
269
+
270
+ def assert_provenance_epoch(self, expected_epoch_id: str) -> None:
271
+ """Fail if another catalog epoch has replaced the caller's active epoch."""
272
+
273
+ active_epoch_id = self.provenance_epoch_id()
274
+ if active_epoch_id != expected_epoch_id:
275
+ raise ProvenanceEpochChangedError(
276
+ "The record catalog provenance epoch changed during this operation; "
277
+ "discard the stale result and start a new Reader invocation."
278
+ )
279
+
280
+ def bind_provenance_epoch(self, expected_epoch_id: str) -> None:
281
+ """Bind subsequent reads and writes to one invocation's catalog epoch."""
282
+
283
+ self.assert_provenance_epoch(expected_epoch_id)
284
+ self._bound_provenance_epoch_id = expected_epoch_id
285
+
286
+ @property
287
+ def provenance_lock(self) -> ProvenanceFileLock:
288
+ """Reentrant writer lease shared by operations, catalog commits, and ledger appends."""
289
+
290
+ return self._catalog_lock
291
+
292
+ @contextmanager
293
+ def _catalog_lock_scope(self, *, operation: str) -> Iterator[None]:
294
+ """Normalize lock failures without swallowing errors from the protected operation."""
295
+
296
+ with provenance_lock_scope(
297
+ self._catalog_lock,
298
+ acquire_error=RecordError(f"Could not {operation}: the record catalog provenance lock is unavailable"),
299
+ release_error=RecordError(f"Could not {operation}: the record catalog provenance lock could not release"),
300
+ release_note="Reader also could not release the record catalog lock",
301
+ ):
302
+ yield
303
+
304
+ def assert_catalog_snapshot(self, *, provenance_epoch_id: str, catalog_digest: str) -> None:
305
+ """Fail unless the catalog still matches a previously read snapshot."""
306
+
307
+ with self._catalog_lock_scope(operation="validate the record catalog snapshot"):
308
+ self.assert_provenance_epoch(provenance_epoch_id)
309
+ if digest_json(self._read_catalog()) != catalog_digest:
310
+ raise RecordError("The record catalog changed concurrently; retry from the current catalog state.")
311
+
312
+ def _source_experiment_resolver(self) -> SourceExperimentResolver:
313
+ experiment_catalog: ExperimentCatalog | None = None
314
+
315
+ def resolve(experiment_id: str):
316
+ nonlocal experiment_catalog
317
+ if experiment_catalog is None:
318
+ experiment_catalog = ExperimentCatalog.from_experiment_root(self.experiment_root)
319
+ return experiment_catalog.resolve(experiment_id)
320
+
321
+ return resolve
322
+
323
+ def _source_identity_resolver(self) -> SourceExperimentResolver:
324
+ """Validate serialized source identity without requiring its live workspace."""
325
+
326
+ def preserve(experiment_id: str) -> ExperimentLocation:
327
+ if Path(experiment_id).name != experiment_id or experiment_id in {".", ".."}:
328
+ raise RecordError(f"Source experiment id must be one safe path segment: {experiment_id!r}")
329
+ return ExperimentLocation(
330
+ id=experiment_id,
331
+ root=self.experiment_root,
332
+ config_path=self.experiment_root / "config.yaml",
333
+ outputs_dir=self.root,
334
+ )
335
+
336
+ return preserve
337
+
338
+ def _read_catalog(
339
+ self,
340
+ *,
341
+ source_experiment_resolver: SourceExperimentResolver | None = None,
342
+ materialize_config_digest: str | None = None,
343
+ materialize_record_ids: frozenset[str] | None = None,
344
+ ) -> dict[str, Any]:
345
+ self._validate_catalog_roots()
346
+ if self.records_path.is_symlink():
347
+ raise RecordError("records.json must not be a symlink")
348
+ if not self.records_path.exists():
349
+ raise RecordError("records.json is missing")
350
+ descriptor = -1
351
+ try:
352
+ flags = os.O_RDONLY | getattr(os, "O_CLOEXEC", 0) | getattr(os, "O_NOFOLLOW", 0)
353
+ flags |= getattr(os, "O_NONBLOCK", 0)
354
+ descriptor = os.open(self.records_path, flags)
355
+ catalog_stat = os.fstat(descriptor)
356
+ if not stat.S_ISREG(catalog_stat.st_mode) or catalog_stat.st_nlink != 1:
357
+ raise RecordError("records.json must be a regular file with a single link")
358
+ with os.fdopen(descriptor, mode="r", encoding="utf-8") as stream:
359
+ descriptor = -1
360
+ content = stream.read()
361
+ except UnicodeError as exc:
362
+ raise RecordError(f"records.json is not valid UTF-8: {exc}") from exc
363
+ except RecordError:
364
+ raise
365
+ except OSError as exc:
366
+ raise RecordError(f"Could not read records.json: {exc}") from exc
367
+ finally:
368
+ if descriptor >= 0:
369
+ os.close(descriptor)
370
+ try:
371
+ data = json.loads(content)
372
+ except json.JSONDecodeError as exc:
373
+ raise RecordError(f"records.json is not valid JSON: {exc}") from exc
374
+ if not isinstance(data, dict):
375
+ raise RecordError("records.json must be a JSON object")
376
+ schema_version = data.get("schema_version")
377
+ if schema_version != RECORD_CATALOG_SCHEMA_VERSION:
378
+ raise RecordError(
379
+ f"records.json schema_version must be {RECORD_CATALOG_SCHEMA_VERSION} (got {schema_version!r})"
380
+ )
381
+ if set(data) != _RECORD_CATALOG_FIELDS:
382
+ raise RecordError(
383
+ "records.json must contain exactly schema_version, provenance_epoch_id, latest, and history"
384
+ )
385
+ if not _is_canonical_uuid4(data.get("provenance_epoch_id")):
386
+ raise RecordError("records.json provenance_epoch_id must be a canonical UUID4")
387
+ if "latest" not in data or "history" not in data:
388
+ raise RecordError("records.json must include 'latest' and 'history' objects")
389
+ if not isinstance(data["latest"], dict) or not isinstance(data["history"], dict):
390
+ raise RecordError("records.json 'latest' and 'history' must be JSON objects")
391
+ selected_record_ids = _selected_latest_record_ids(
392
+ data["latest"],
393
+ config_digest=materialize_config_digest,
394
+ record_ids=materialize_record_ids,
395
+ )
396
+ self._validate_catalog_lineage(
397
+ data,
398
+ source_experiment_resolver=source_experiment_resolver or self._source_experiment_resolver(),
399
+ materialize_latest_ids=selected_record_ids,
400
+ materialize_history=materialize_config_digest is None and materialize_record_ids is None,
401
+ )
402
+ if (
403
+ self._bound_provenance_epoch_id is not None
404
+ and data["provenance_epoch_id"] != self._bound_provenance_epoch_id
405
+ ):
406
+ raise ProvenanceEpochChangedError(
407
+ "The record catalog provenance epoch changed during this operation; "
408
+ "discard the stale result and start a new Reader invocation."
409
+ )
410
+ return data
411
+
412
+ def _validate_catalog_lineage(
413
+ self,
414
+ catalog: dict[str, Any],
415
+ *,
416
+ source_experiment_resolver: SourceExperimentResolver,
417
+ materialize_latest_ids: frozenset[str],
418
+ materialize_history: bool,
419
+ ) -> None:
420
+ latest = catalog["latest"]
421
+ history = catalog["history"]
422
+ latest_ids = set(latest)
423
+ history_ids = set(history)
424
+ if latest_ids != history_ids:
425
+ missing_history = sorted(latest_ids - history_ids)
426
+ orphan_history = sorted(history_ids - latest_ids)
427
+ details = []
428
+ if missing_history:
429
+ details.append("latest records missing history: " + ", ".join(missing_history))
430
+ if orphan_history:
431
+ details.append("history records missing latest: " + ", ".join(orphan_history))
432
+ raise RecordError("records.json latest/history lineage is inconsistent: " + "; ".join(details))
433
+ identity_resolver = self._source_identity_resolver()
434
+ for record_id in sorted(latest_ids):
435
+ revisions = history[record_id]
436
+ if not isinstance(revisions, list):
437
+ raise RecordError(f"records.json history for {record_id!r} must be a list")
438
+ if not revisions:
439
+ raise RecordError(f"records.json history for latest record {record_id!r} must be non-empty")
440
+ if not materialize_history:
441
+ for payload in revisions:
442
+ self._materialize(
443
+ record_id,
444
+ payload,
445
+ source_experiment_resolver=identity_resolver,
446
+ )
447
+ self._materialize(
448
+ record_id,
449
+ latest[record_id],
450
+ source_experiment_resolver=identity_resolver,
451
+ )
452
+ if materialize_history:
453
+ for payload in revisions:
454
+ self._materialize(
455
+ record_id,
456
+ payload,
457
+ source_experiment_resolver=source_experiment_resolver,
458
+ )
459
+ if record_id in materialize_latest_ids:
460
+ self._materialize(
461
+ record_id,
462
+ latest[record_id],
463
+ source_experiment_resolver=source_experiment_resolver,
464
+ )
465
+ if revisions[-1] != latest[record_id]:
466
+ raise RecordError(
467
+ f"records.json history for {record_id!r} must end with the exact latest record payload"
468
+ )
469
+
470
+ def _write_catalog(
471
+ self,
472
+ payload: dict[str, Any],
473
+ *,
474
+ expected_provenance_epoch_id: str | None = None,
475
+ expected_catalog_digest: str | None = None,
476
+ create_only: bool = False,
477
+ ) -> None:
478
+ self._validate_catalog_roots()
479
+ self.manifests_dir.mkdir(parents=True, exist_ok=True)
480
+ staged_path: Path | None = None
481
+ try:
482
+ with NamedTemporaryFile(
483
+ mode="w",
484
+ encoding="utf-8",
485
+ dir=self.manifests_dir,
486
+ prefix=f".{self.records_path.name}.",
487
+ suffix=".tmp",
488
+ delete=False,
489
+ ) as staged:
490
+ staged_path = Path(staged.name)
491
+ json.dump(payload, staged, indent=2, sort_keys=True)
492
+ staged.flush()
493
+ os.fsync(staged.fileno())
494
+ try:
495
+ with self._catalog_lock_scope(operation="commit the record catalog"):
496
+ self._validate_catalog_roots()
497
+ if create_only and self.records_path.exists():
498
+ return
499
+ if expected_provenance_epoch_id is not None:
500
+ self.assert_provenance_epoch(expected_provenance_epoch_id)
501
+ if expected_catalog_digest is not None:
502
+ current_digest = digest_json(self._read_catalog())
503
+ if current_digest != expected_catalog_digest:
504
+ raise RecordError(
505
+ "The record catalog changed concurrently; retry from the current catalog state."
506
+ )
507
+ staged_path.replace(self.records_path)
508
+ except OSError as exc:
509
+ raise RecordError(f"Could not atomically replace records.json: {exc}") from exc
510
+ finally:
511
+ if staged_path is not None:
512
+ staged_path.unlink(missing_ok=True)
513
+
514
+ def _materialize(
515
+ self,
516
+ record_id: str,
517
+ payload: dict[str, Any],
518
+ *,
519
+ source_experiment_resolver: SourceExperimentResolver | None = None,
520
+ ) -> DataFrameArtifactRecord | FileBundleRecord:
521
+ record = record_from_dict(
522
+ payload,
523
+ outputs_dir=self.root,
524
+ experiment_root=self.experiment_root,
525
+ source_experiment_resolver=source_experiment_resolver or self._source_experiment_resolver(),
526
+ )
527
+ if record.record_id != record_id:
528
+ raise RecordError(
529
+ f"records.json entry key {record_id!r} does not match payload record_id {record.record_id!r}"
530
+ )
531
+ return record
532
+
533
+ def capture_inputs(
534
+ self,
535
+ inputs: Iterable[ProvenanceInput],
536
+ *,
537
+ resolved_inputs: Mapping[str, Any] | None = None,
538
+ ) -> tuple[RecordInputEvidence, ...]:
539
+ """Bind immutable evidence to the exact inputs resolved for one computation."""
540
+
541
+ resolved_by_label = dict(resolved_inputs or {})
542
+ from .sources import ResolvedSourceRecord, SourceRecordCollection # noqa: PLC0415
543
+
544
+ resolved_sources: dict[SourceRecordRef, ResolvedSourceRecord] = {}
545
+ for value in resolved_by_label.values():
546
+ if isinstance(value, ResolvedSourceRecord):
547
+ candidates = (value,)
548
+ elif isinstance(value, SourceRecordCollection):
549
+ candidates = value.records
550
+ else:
551
+ continue
552
+ for candidate in candidates:
553
+ previous = resolved_sources.get(candidate.ref)
554
+ if previous is not None and previous.revision_digest != candidate.revision_digest:
555
+ raise RecordError(
556
+ f"Resolved source record {candidate.ref.experiment_id}:{candidate.ref.record_id} "
557
+ "has conflicting revisions"
558
+ )
559
+ resolved_sources[candidate.ref] = candidate
560
+
561
+ evidence: list[RecordInputEvidence] = []
562
+ for item in inputs:
563
+ if isinstance(item.ref, RecordRef):
564
+ upstream = resolved_by_label.get(item.label)
565
+ if upstream is None:
566
+ upstream = self.latest_record(item.ref.record_id)
567
+ if upstream is None:
568
+ raise RecordError(
569
+ f"Input record {item.ref.record_id!r} is missing; produce it before persisting this record."
570
+ )
571
+ if not isinstance(upstream, (DataFrameArtifactRecord, FileBundleRecord)):
572
+ raise RecordError(f"Resolved input {item.label!r} must be a Reader record")
573
+ if upstream.record_id != item.ref.record_id:
574
+ raise RecordError(
575
+ f"Resolved input {item.label!r} is record {upstream.record_id!r}, "
576
+ f"expected {item.ref.record_id!r}"
577
+ )
578
+ verify_record_artifact_integrity(upstream, outputs_dir=self.root)
579
+ evidence.append(
580
+ RecordInputEvidence(
581
+ label=item.label,
582
+ ref=item.ref,
583
+ discovery_policy="record",
584
+ record_revision_digest=record_revision_digest(upstream, outputs_dir=self.root),
585
+ )
586
+ )
587
+ continue
588
+ if isinstance(item.ref, SourceRecordRef):
589
+ from .sources import resolve_source_record # noqa: PLC0415
590
+
591
+ upstream = resolved_by_label.get(item.label)
592
+ if upstream is None:
593
+ upstream = resolved_sources.get(item.ref)
594
+ if upstream is None:
595
+ upstream = resolve_source_record(item.ref, contracts=self.contracts)
596
+ if not isinstance(upstream, ResolvedSourceRecord):
597
+ raise RecordError(f"Resolved input {item.label!r} must be a Reader source record")
598
+ if upstream.ref != item.ref:
599
+ raise RecordError(
600
+ f"Resolved input {item.label!r} is source record "
601
+ f"{upstream.ref.experiment_id}:{upstream.ref.record_id}, expected "
602
+ f"{item.ref.experiment_id}:{item.ref.record_id}"
603
+ )
604
+ upstream.verify_artifact_integrity()
605
+ evidence.append(
606
+ RecordInputEvidence(
607
+ label=item.label,
608
+ ref=item.ref,
609
+ discovery_policy="source_record",
610
+ record_revision_digest=upstream.revision_digest,
611
+ )
612
+ )
613
+ continue
614
+ if isinstance(item.ref, ResourceRef):
615
+ default_policy = "declared_resource"
616
+ elif isinstance(item.ref, FileRef):
617
+ default_policy = "declared_file"
618
+ else:
619
+ raise RecordError(f"Unsupported input reference for {item.label!r}")
620
+ evidence.append(
621
+ RecordInputEvidence(
622
+ label=item.label,
623
+ ref=item.ref,
624
+ discovery_policy=item.discovery_policy or default_policy,
625
+ artifact=capture_artifact_evidence(item.ref.path, root=self.experiment_root),
626
+ )
627
+ )
628
+ return tuple(evidence)
629
+
630
+ def _require_captured_inputs(self, inputs: Iterable[RecordInputEvidence]) -> tuple[RecordInputEvidence, ...]:
631
+ captured = tuple(inputs)
632
+ if any(not isinstance(item, RecordInputEvidence) for item in captured):
633
+ raise RecordError(
634
+ "Record persistence requires pre-captured RecordInputEvidence; "
635
+ "capture inputs before computation with RecordStore.capture_inputs()."
636
+ )
637
+ self._assert_captured_inputs_current(captured)
638
+ return captured
639
+
640
+ def _assert_captured_inputs_current(self, captured: Iterable[RecordInputEvidence]) -> None:
641
+ for item in captured:
642
+ if isinstance(item.ref, RecordRef):
643
+ current = self.latest_record(item.ref.record_id)
644
+ if current is None:
645
+ raise RecordError(
646
+ f"Input record {item.ref.record_id!r} changed after input evidence was captured: "
647
+ "the current revision is missing."
648
+ )
649
+ current_revision = record_revision_digest(current, outputs_dir=self.root)
650
+ if current_revision != item.record_revision_digest:
651
+ raise RecordError(f"Input record {item.ref.record_id!r} changed after input evidence was captured.")
652
+ try:
653
+ verify_record_artifact_integrity(current, outputs_dir=self.root)
654
+ except RecordError as exc:
655
+ raise RecordError(
656
+ f"Input record {item.ref.record_id!r} changed after input evidence was captured: {exc}"
657
+ ) from exc
658
+ continue
659
+ if isinstance(item.ref, SourceRecordRef):
660
+ from .sources import resolve_source_record # noqa: PLC0415
661
+
662
+ try:
663
+ current = resolve_source_record(item.ref, contracts=self.contracts)
664
+ current.verify_artifact_integrity()
665
+ except RecordError as exc:
666
+ raise RecordError(
667
+ f"Source record {item.ref.experiment_id}:{item.ref.record_id} changed after input "
668
+ f"evidence was captured: {exc}"
669
+ ) from exc
670
+ if current.revision_digest != item.record_revision_digest:
671
+ raise RecordError(
672
+ f"Source record {item.ref.experiment_id}:{item.ref.record_id} changed after input "
673
+ "evidence was captured."
674
+ )
675
+ continue
676
+ try:
677
+ current_artifact = capture_artifact_evidence(item.ref.path, root=self.experiment_root)
678
+ except (OSError, RecordError) as exc:
679
+ raise RecordError(
680
+ f"Input file for {item.label!r} changed after input evidence was captured: {exc}"
681
+ ) from exc
682
+ if current_artifact != item.artifact:
683
+ raise RecordError(f"Input file for {item.label!r} changed after input evidence was captured.")
684
+
685
+ def iter_latest_records(
686
+ self,
687
+ *,
688
+ kind: WorkbenchRecordKind | None = None,
689
+ producer_kind: WorkbenchProducerKind | None = None,
690
+ ) -> tuple[DataFrameArtifactRecord | FileBundleRecord, ...]:
691
+ source_experiment_resolver = self._source_experiment_resolver()
692
+ catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
693
+ out: list[DataFrameArtifactRecord | FileBundleRecord] = []
694
+ for record_id, payload in sorted(catalog["latest"].items()):
695
+ record = self._materialize(
696
+ record_id,
697
+ payload,
698
+ source_experiment_resolver=source_experiment_resolver,
699
+ )
700
+ if kind is not None and record.kind != kind:
701
+ continue
702
+ if producer_kind is not None and record.producer.kind != producer_kind:
703
+ continue
704
+ out.append(record)
705
+ return tuple(out)
706
+
707
+ def record_history(self, record_id: str) -> tuple[DataFrameArtifactRecord | FileBundleRecord, ...]:
708
+ source_experiment_resolver = self._source_experiment_resolver()
709
+ catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
710
+ history = catalog["history"].get(record_id, [])
711
+ if not isinstance(history, list):
712
+ raise RecordError(f"records.json history for {record_id!r} must be a list")
713
+ return tuple(
714
+ self._materialize(
715
+ record_id,
716
+ payload,
717
+ source_experiment_resolver=source_experiment_resolver,
718
+ )
719
+ for payload in history
720
+ )
721
+
722
+ def revision_counts(self, record_ids: Iterable[str] | None = None) -> dict[str, int]:
723
+ catalog = self._read_catalog()
724
+ requested = None if record_ids is None else {str(record_id) for record_id in record_ids}
725
+ counts: dict[str, int] = {}
726
+ for record_id, history in catalog["history"].items():
727
+ if requested is not None and record_id not in requested:
728
+ continue
729
+ if not isinstance(history, list):
730
+ raise RecordError(f"records.json history for {record_id!r} must be a list")
731
+ counts[record_id] = len(history)
732
+ if requested is not None:
733
+ for record_id in requested:
734
+ counts.setdefault(record_id, 0)
735
+ return counts
736
+
737
+ def latest_dataframe(self, record_id: str) -> DataFrameArtifactRecord | None:
738
+ source_experiment_resolver = self._source_experiment_resolver()
739
+ catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
740
+ payload = catalog["latest"].get(record_id)
741
+ if payload is None:
742
+ return None
743
+ record = self._materialize(
744
+ record_id,
745
+ payload,
746
+ source_experiment_resolver=source_experiment_resolver,
747
+ )
748
+ if not isinstance(record, DataFrameArtifactRecord):
749
+ raise RecordError(f"Record {record_id!r} exists but is not a dataframe artifact")
750
+ return record
751
+
752
+ def latest_record(self, record_id: str) -> DataFrameArtifactRecord | FileBundleRecord | None:
753
+ source_experiment_resolver = self._source_experiment_resolver()
754
+ catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
755
+ payload = catalog["latest"].get(record_id)
756
+ if payload is None:
757
+ return None
758
+ return self._materialize(
759
+ record_id,
760
+ payload,
761
+ source_experiment_resolver=source_experiment_resolver,
762
+ )
763
+
764
+ def read_dataframe(self, record_id: str) -> DataFrameArtifactRecord:
765
+ record = self.latest_dataframe(record_id)
766
+ if record is None:
767
+ raise RecordError(f"Dataframe record '{record_id}' is missing; produce it in an earlier step.")
768
+ return record
769
+
770
+ def read_record(self, record_id: str) -> DataFrameArtifactRecord | FileBundleRecord:
771
+ record = self.latest_record(record_id)
772
+ if record is None:
773
+ raise RecordError(f"Record '{record_id}' is missing; produce it in an earlier step.")
774
+ return record
775
+
776
+ def _revision_dir(self, step_dir: Path) -> Path:
777
+ index = 1
778
+ while True:
779
+ candidate = step_dir if index == 1 else step_dir.with_name(f"{step_dir.name}__r{index}")
780
+ if not candidate.exists():
781
+ return candidate
782
+ index += 1
783
+
784
+ def persist_dataframe(
785
+ self,
786
+ *,
787
+ producer_id: str,
788
+ producer_plugin: str,
789
+ out_name: str,
790
+ record_id: str,
791
+ df: pd.DataFrame,
792
+ contract_id: ContractId,
793
+ inputs: Iterable[RecordInputEvidence],
794
+ config_digest: str,
795
+ producer_config_digest: str | None = None,
796
+ code_digest: str | None = None,
797
+ build_identity: BuildIdentity | None = None,
798
+ producer_kind: WorkbenchProducerKind = "pipeline",
799
+ source_recipe: RecipeSource | None = None,
800
+ ) -> DataFrameArtifactRecord:
801
+ self.contracts.validate(df, contract_id=contract_id, where=f"{producer_id}:{out_name}")
802
+ producer = RecordProducer(
803
+ kind=producer_kind,
804
+ id=producer_id,
805
+ plugin=producer_plugin,
806
+ source_recipe=(
807
+ RecordRecipeSource(recipe=source_recipe.recipe, with_=dict(source_recipe.with_ or {}))
808
+ if source_recipe is not None
809
+ else None
810
+ ),
811
+ )
812
+ record_inputs = self._require_captured_inputs(inputs)
813
+ effective_build_identity = build_identity or current_build_identity()
814
+ effective_code_digest = code_digest or effective_build_identity.source_digest
815
+ effective_producer_config_digest = producer_config_digest or config_digest
816
+ self.ensure_layout()
817
+ base_name = f"{producer_id}.{producer_plugin.replace('/', '_')}"
818
+ base_step_dir = self.artifacts_dir / base_name
819
+ source_experiment_resolver = self._source_experiment_resolver()
820
+ catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
821
+ provenance_epoch_id = str(catalog["provenance_epoch_id"])
822
+ catalog_digest = digest_json(catalog)
823
+ previous_payload = catalog["latest"].get(record_id)
824
+ previous_record = None
825
+ if previous_payload is not None:
826
+ previous_record = self._materialize(
827
+ record_id,
828
+ previous_payload,
829
+ source_experiment_resolver=source_experiment_resolver,
830
+ )
831
+ if not isinstance(previous_record, DataFrameArtifactRecord):
832
+ raise RecordError(
833
+ f"Record id {record_id!r} is already used by a non-dataframe record; choose a unique id."
834
+ )
835
+ filename = f"{out_name}.parquet"
836
+ staged_path: Path | None = None
837
+ revision_dir: Path | None = None
838
+ data_path: Path | None = None
839
+ try:
840
+ with NamedTemporaryFile(
841
+ dir=self.artifacts_dir,
842
+ prefix=f".{base_name}.",
843
+ suffix=".parquet",
844
+ delete=False,
845
+ ) as staged:
846
+ staged_path = Path(staged.name)
847
+ df.to_parquet(staged_path, index=False)
848
+ content_digest = sha256_file(staged_path)
849
+ if (
850
+ previous_record is not None
851
+ and previous_record.config_digest == config_digest
852
+ and previous_record.content_digest == content_digest
853
+ and previous_record.contract_id == contract_id
854
+ and previous_record.code_digest == effective_code_digest
855
+ and previous_record.producer_config_digest == effective_producer_config_digest
856
+ and previous_record.build_identity == effective_build_identity
857
+ and previous_record.producer == producer
858
+ and previous_record.inputs == record_inputs
859
+ and previous_record.path.name == filename
860
+ ):
861
+ previous_record.verify_content_digest()
862
+ self.assert_catalog_snapshot(
863
+ provenance_epoch_id=provenance_epoch_id,
864
+ catalog_digest=catalog_digest,
865
+ )
866
+ return previous_record
867
+
868
+ revision_dir = self._revision_dir(base_step_dir)
869
+ revision_dir.mkdir(parents=True)
870
+ data_path = revision_dir / filename
871
+ staged_path.replace(data_path)
872
+ staged_path = None
873
+ record = DataFrameArtifactRecord(
874
+ record_id=record_id,
875
+ kind="dataframe_artifact",
876
+ producer=producer,
877
+ created_at=datetime.now(UTC).isoformat(),
878
+ inputs=record_inputs,
879
+ config_digest=config_digest,
880
+ contract_id=contract_id,
881
+ path=data_path,
882
+ content_digest=content_digest,
883
+ code_digest=effective_code_digest,
884
+ producer_config_digest=effective_producer_config_digest,
885
+ build_identity=effective_build_identity,
886
+ size_bytes=data_path.stat().st_size,
887
+ )
888
+ payload = record_to_dict(record, outputs_dir=self.root)
889
+ self._assert_captured_inputs_current(record_inputs)
890
+ catalog["latest"][record_id] = payload
891
+ catalog["history"].setdefault(record_id, []).append(payload)
892
+ self._write_catalog(
893
+ catalog,
894
+ expected_provenance_epoch_id=provenance_epoch_id,
895
+ expected_catalog_digest=catalog_digest,
896
+ )
897
+ return record
898
+ except Exception:
899
+ if data_path is not None:
900
+ data_path.unlink(missing_ok=True)
901
+ if revision_dir is not None:
902
+ with suppress(OSError):
903
+ revision_dir.rmdir()
904
+ raise
905
+ finally:
906
+ if staged_path is not None:
907
+ staged_path.unlink(missing_ok=True)
908
+
909
+ def append_file_bundle(
910
+ self,
911
+ *,
912
+ producer_kind: WorkbenchProducerKind,
913
+ producer_id: str,
914
+ producer_plugin: str,
915
+ record_id: str,
916
+ inputs: Iterable[RecordInputEvidence],
917
+ config_digest: str,
918
+ producer_config_digest: str | None = None,
919
+ files: list[Path],
920
+ description: str,
921
+ path_descriptions: tuple[PathDescription, ...] = (),
922
+ source_recipe: RecipeSource | None = None,
923
+ build_identity: BuildIdentity | None = None,
924
+ ) -> FileBundleRecord:
925
+ producer = RecordProducer(
926
+ kind=producer_kind,
927
+ id=producer_id,
928
+ plugin=producer_plugin,
929
+ source_recipe=(
930
+ RecordRecipeSource(recipe=source_recipe.recipe, with_=dict(source_recipe.with_ or {}))
931
+ if source_recipe is not None
932
+ else None
933
+ ),
934
+ )
935
+ return self._append_file_bundle_record(
936
+ producer=producer,
937
+ record_id=record_id,
938
+ inputs=inputs,
939
+ config_digest=config_digest,
940
+ producer_config_digest=producer_config_digest,
941
+ files=files,
942
+ description=description,
943
+ path_descriptions=path_descriptions,
944
+ build_identity=build_identity,
945
+ )
946
+
947
+ def _append_file_bundle_record(
948
+ self,
949
+ *,
950
+ producer: RecordProducer,
951
+ record_id: str,
952
+ inputs: Iterable[RecordInputEvidence],
953
+ config_digest: str,
954
+ producer_config_digest: str | None,
955
+ files: list[Path],
956
+ description: str,
957
+ path_descriptions: tuple[PathDescription, ...],
958
+ build_identity: BuildIdentity | None,
959
+ ) -> FileBundleRecord:
960
+ normalized_files = tuple(sorted(dict.fromkeys(files), key=str))
961
+ # Validate bundle semantics before touching the filesystem so malformed
962
+ # descriptions and path mappings fail independently of artifact state.
963
+ description, path_descriptions = normalize_file_bundle_metadata(
964
+ producer_kind=producer.kind,
965
+ files=normalized_files,
966
+ description=description,
967
+ path_descriptions=path_descriptions,
968
+ )
969
+ materialized_paths = tuple(path if path.is_absolute() else self.root / path for path in normalized_files)
970
+ missing = [path for path in materialized_paths if not path.exists()]
971
+ if missing:
972
+ raise RecordError(
973
+ f"File-bundle record {record_id!r} cannot persist missing files: "
974
+ + ", ".join(str(path) for path in missing)
975
+ )
976
+ non_files = [path for path in materialized_paths if not path.is_file()]
977
+ if non_files:
978
+ raise RecordError(
979
+ f"File-bundle record {record_id!r} cannot persist non-file paths: "
980
+ + ", ".join(str(path) for path in non_files)
981
+ )
982
+ file_evidence = tuple(capture_artifact_evidence(path, root=self.root) for path in materialized_paths)
983
+ effective_build_identity = build_identity or current_build_identity()
984
+ record_inputs = self._require_captured_inputs(inputs)
985
+ self.ensure_layout()
986
+ source_experiment_resolver = self._source_experiment_resolver()
987
+ catalog = self._read_catalog(source_experiment_resolver=source_experiment_resolver)
988
+ provenance_epoch_id = str(catalog["provenance_epoch_id"])
989
+ catalog_digest = digest_json(catalog)
990
+ previous_payload = catalog["latest"].get(record_id)
991
+ if previous_payload is not None:
992
+ previous_record = self._materialize(
993
+ record_id,
994
+ previous_payload,
995
+ source_experiment_resolver=source_experiment_resolver,
996
+ )
997
+ if isinstance(previous_record, DataFrameArtifactRecord):
998
+ raise RecordError(f"Record id {record_id!r} is already used by a dataframe record; choose a unique id.")
999
+ record = FileBundleRecord(
1000
+ record_id=record_id,
1001
+ kind="file_bundle",
1002
+ producer=producer,
1003
+ created_at=datetime.now(UTC).isoformat(),
1004
+ inputs=record_inputs,
1005
+ config_digest=config_digest,
1006
+ files=normalized_files,
1007
+ description=description,
1008
+ path_descriptions=path_descriptions,
1009
+ file_evidence=file_evidence,
1010
+ producer_config_digest=producer_config_digest or config_digest,
1011
+ build_identity=effective_build_identity,
1012
+ )
1013
+ payload = record_to_dict(record, outputs_dir=self.root)
1014
+ self._assert_captured_inputs_current(record_inputs)
1015
+ catalog["latest"][record_id] = payload
1016
+ catalog["history"].setdefault(record_id, []).append(payload)
1017
+ self._write_catalog(
1018
+ catalog,
1019
+ expected_provenance_epoch_id=provenance_epoch_id,
1020
+ expected_catalog_digest=catalog_digest,
1021
+ )
1022
+ return record