reader-workbench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (293) hide show
  1. reader_workbench/__init__.py +22 -0
  2. reader_workbench/__main__.py +4 -0
  3. reader_workbench/_version.py +17 -0
  4. reader_workbench/api/__init__.py +74 -0
  5. reader_workbench/api/_record_reads.py +75 -0
  6. reader_workbench/api/artifacts.py +79 -0
  7. reader_workbench/api/facade.py +538 -0
  8. reader_workbench/api/models.py +285 -0
  9. reader_workbench/api/notebooks.py +63 -0
  10. reader_workbench/contracts/__init__.py +18 -0
  11. reader_workbench/contracts/builtins/__init__.py +36 -0
  12. reader_workbench/contracts/builtins/cytometry.py +140 -0
  13. reader_workbench/contracts/builtins/four_state_event_window.py +214 -0
  14. reader_workbench/contracts/builtins/generic.py +21 -0
  15. reader_workbench/contracts/builtins/logic.py +149 -0
  16. reader_workbench/contracts/builtins/plate_reader.py +47 -0
  17. reader_workbench/contracts/catalog.py +257 -0
  18. reader_workbench/contracts/model.py +109 -0
  19. reader_workbench/domains/__init__.py +1 -0
  20. reader_workbench/domains/cytometry/__init__.py +3 -0
  21. reader_workbench/domains/cytometry/analysis/__init__.py +29 -0
  22. reader_workbench/domains/cytometry/analysis/events.py +182 -0
  23. reader_workbench/domains/cytometry/analysis/gating.py +175 -0
  24. reader_workbench/domains/cytometry/analysis/workflow.py +248 -0
  25. reader_workbench/domains/cytometry/io/__init__.py +3 -0
  26. reader_workbench/domains/cytometry/io/fcs.py +135 -0
  27. reader_workbench/domains/cytometry/plots/__init__.py +5 -0
  28. reader_workbench/domains/cytometry/plots/diagnostic.py +155 -0
  29. reader_workbench/domains/logic/__init__.py +3 -0
  30. reader_workbench/domains/logic/crosstalk/__init__.py +3 -0
  31. reader_workbench/domains/logic/crosstalk/pairs.py +661 -0
  32. reader_workbench/domains/logic/four_state_vector/__init__.py +7 -0
  33. reader_workbench/domains/logic/four_state_vector/builder.py +321 -0
  34. reader_workbench/domains/logic/four_state_vector/collection/__init__.py +20 -0
  35. reader_workbench/domains/logic/four_state_vector/collection/checks.py +49 -0
  36. reader_workbench/domains/logic/four_state_vector/collection/constants.py +29 -0
  37. reader_workbench/domains/logic/four_state_vector/collection/model.py +19 -0
  38. reader_workbench/domains/logic/four_state_vector/collection/render.py +383 -0
  39. reader_workbench/domains/logic/four_state_vector/collection/sources.py +185 -0
  40. reader_workbench/domains/logic/four_state_vector/config.py +214 -0
  41. reader_workbench/domains/logic/four_state_vector/diagnostic.py +361 -0
  42. reader_workbench/domains/logic/four_state_vector/heatmap.py +86 -0
  43. reader_workbench/domains/logic/four_state_vector/math.py +191 -0
  44. reader_workbench/domains/logic/four_state_vector/reference.py +85 -0
  45. reader_workbench/domains/logic/four_state_vector/selection.py +228 -0
  46. reader_workbench/domains/logic/four_state_vector/treatment_semantics.py +51 -0
  47. reader_workbench/domains/logic/four_state_vector/validation.py +19 -0
  48. reader_workbench/domains/logic/logic_symmetry/__init__.py +3 -0
  49. reader_workbench/domains/logic/logic_symmetry/encodings.py +93 -0
  50. reader_workbench/domains/logic/logic_symmetry/extract_corners.py +192 -0
  51. reader_workbench/domains/logic/logic_symmetry/main.py +236 -0
  52. reader_workbench/domains/logic/logic_symmetry/metrics.py +96 -0
  53. reader_workbench/domains/logic/logic_symmetry/overlay.py +129 -0
  54. reader_workbench/domains/logic/logic_symmetry/prep.py +138 -0
  55. reader_workbench/domains/logic/logic_symmetry/render.py +356 -0
  56. reader_workbench/domains/logic/treatment_columns.py +42 -0
  57. reader_workbench/domains/plate_reader/__init__.py +1 -0
  58. reader_workbench/domains/plate_reader/analysis/__init__.py +14 -0
  59. reader_workbench/domains/plate_reader/analysis/fold_change.py +474 -0
  60. reader_workbench/domains/plate_reader/analysis/four_state_event_window/__init__.py +21 -0
  61. reader_workbench/domains/plate_reader/analysis/four_state_event_window/aggregation.py +191 -0
  62. reader_workbench/domains/plate_reader/analysis/four_state_event_window/contract_fields.py +50 -0
  63. reader_workbench/domains/plate_reader/analysis/four_state_event_window/contracts.py +320 -0
  64. reader_workbench/domains/plate_reader/analysis/four_state_event_window/design_dispositions.py +54 -0
  65. reader_workbench/domains/plate_reader/analysis/four_state_event_window/disposition_records.py +140 -0
  66. reader_workbench/domains/plate_reader/analysis/four_state_event_window/event_sensitivity.py +27 -0
  67. reader_workbench/domains/plate_reader/analysis/four_state_event_window/materialize.py +274 -0
  68. reader_workbench/domains/plate_reader/analysis/four_state_event_window/observation_resampling.py +97 -0
  69. reader_workbench/domains/plate_reader/analysis/four_state_event_window/reduction.py +62 -0
  70. reader_workbench/domains/plate_reader/analysis/four_state_event_window/seeds.py +15 -0
  71. reader_workbench/domains/plate_reader/analysis/four_state_event_window/sources.py +308 -0
  72. reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusion_validation.py +94 -0
  73. reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusions.py +54 -0
  74. reader_workbench/domains/plate_reader/analysis/timepoints.py +76 -0
  75. reader_workbench/domains/plate_reader/io/__init__.py +6 -0
  76. reader_workbench/domains/plate_reader/io/sample_map.py +65 -0
  77. reader_workbench/domains/plate_reader/io/synergy_h1/__init__.py +6 -0
  78. reader_workbench/domains/plate_reader/io/synergy_h1/_kinetic.py +141 -0
  79. reader_workbench/domains/plate_reader/io/synergy_h1/_parser.py +295 -0
  80. reader_workbench/domains/plate_reader/io/synergy_h1/_shared.py +195 -0
  81. reader_workbench/domains/plate_reader/io/synergy_h1/_snapshot.py +130 -0
  82. reader_workbench/domains/plate_reader/ordering.py +59 -0
  83. reader_workbench/domains/plate_reader/plots/__init__.py +15 -0
  84. reader_workbench/domains/plate_reader/plots/_data.py +29 -0
  85. reader_workbench/domains/plate_reader/plots/common.py +346 -0
  86. reader_workbench/domains/plate_reader/plots/distributions.py +324 -0
  87. reader_workbench/domains/plate_reader/plots/dual_reporter_triptych.py +525 -0
  88. reader_workbench/domains/plate_reader/plots/dual_reporter_triptych_render.py +195 -0
  89. reader_workbench/domains/plate_reader/plots/four_state_event_window/__init__.py +26 -0
  90. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic.py +295 -0
  91. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_components.py +149 -0
  92. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_render.py +308 -0
  93. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_style.py +51 -0
  94. reader_workbench/domains/plate_reader/plots/four_state_event_window/schema.py +8 -0
  95. reader_workbench/domains/plate_reader/plots/four_state_event_window/summary.py +140 -0
  96. reader_workbench/domains/plate_reader/plots/grouping.py +53 -0
  97. reader_workbench/domains/plate_reader/plots/panels/__init__.py +12 -0
  98. reader_workbench/domains/plate_reader/plots/panels/snapshot.py +161 -0
  99. reader_workbench/domains/plate_reader/plots/panels/snapshot_data.py +91 -0
  100. reader_workbench/domains/plate_reader/plots/panels/time_series.py +296 -0
  101. reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic.py +435 -0
  102. reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic_render.py +300 -0
  103. reader_workbench/domains/plate_reader/plots/snapshot_barplot/__init__.py +315 -0
  104. reader_workbench/domains/plate_reader/plots/snapshot_barplot/planning.py +168 -0
  105. reader_workbench/domains/plate_reader/plots/snapshot_heatmap/__init__.py +205 -0
  106. reader_workbench/domains/plate_reader/plots/snapshot_heatmap/inputs.py +103 -0
  107. reader_workbench/domains/plate_reader/plots/time_series.py +317 -0
  108. reader_workbench/domains/plate_reader/plots/ts_and_snap/__init__.py +479 -0
  109. reader_workbench/domains/plate_reader/plots/ts_and_snap/planning.py +283 -0
  110. reader_workbench/domains/time_series/__init__.py +29 -0
  111. reader_workbench/domains/time_series/aggregation.py +60 -0
  112. reader_workbench/domains/time_series/contracts.py +368 -0
  113. reader_workbench/domains/time_series/reduction.py +395 -0
  114. reader_workbench/errors.py +55 -0
  115. reader_workbench/maintenance/__init__.py +6 -0
  116. reader_workbench/maintenance/docs.py +335 -0
  117. reader_workbench/maintenance/model.py +28 -0
  118. reader_workbench/maintenance/release.py +39 -0
  119. reader_workbench/maintenance/skills.py +124 -0
  120. reader_workbench/plotting/__init__.py +20 -0
  121. reader_workbench/plotting/mpl.py +56 -0
  122. reader_workbench/plotting/sinks.py +69 -0
  123. reader_workbench/plotting/style.py +175 -0
  124. reader_workbench/plotting/utils.py +27 -0
  125. reader_workbench/plugins/__init__.py +1 -0
  126. reader_workbench/plugins/catalog.py +33 -0
  127. reader_workbench/plugins/export/__init__.py +0 -0
  128. reader_workbench/plugins/export/_paths.py +21 -0
  129. reader_workbench/plugins/export/csv.py +41 -0
  130. reader_workbench/plugins/export/xlsx.py +44 -0
  131. reader_workbench/plugins/ingest/__init__.py +0 -0
  132. reader_workbench/plugins/ingest/_discovery.py +58 -0
  133. reader_workbench/plugins/ingest/discovery_policy.py +66 -0
  134. reader_workbench/plugins/ingest/flow_cytometer.py +139 -0
  135. reader_workbench/plugins/ingest/synergy_h1.py +234 -0
  136. reader_workbench/plugins/manifests/__init__.py +1 -0
  137. reader_workbench/plugins/manifests/export.py +29 -0
  138. reader_workbench/plugins/manifests/ingest.py +29 -0
  139. reader_workbench/plugins/manifests/plot.py +161 -0
  140. reader_workbench/plugins/manifests/transform.py +172 -0
  141. reader_workbench/plugins/manifests/validator.py +18 -0
  142. reader_workbench/plugins/plot/__init__.py +0 -0
  143. reader_workbench/plugins/plot/_shared.py +55 -0
  144. reader_workbench/plugins/plot/cytometry_diagnostic.py +54 -0
  145. reader_workbench/plugins/plot/distributions.py +58 -0
  146. reader_workbench/plugins/plot/dual_reporter_triptych.py +224 -0
  147. reader_workbench/plugins/plot/four_state_event_window_diagnostic.py +133 -0
  148. reader_workbench/plugins/plot/four_state_event_window_summary.py +53 -0
  149. reader_workbench/plugins/plot/four_state_vector_collection.py +45 -0
  150. reader_workbench/plugins/plot/four_state_vector_diagnostic.py +105 -0
  151. reader_workbench/plugins/plot/four_state_vector_heatmap.py +63 -0
  152. reader_workbench/plugins/plot/logic_symmetry.py +56 -0
  153. reader_workbench/plugins/plot/single_reporter_diagnostic.py +279 -0
  154. reader_workbench/plugins/plot/snapshot_barplot.py +65 -0
  155. reader_workbench/plugins/plot/snapshot_heatmap.py +104 -0
  156. reader_workbench/plugins/plot/time_series.py +114 -0
  157. reader_workbench/plugins/plot/ts_and_snap.py +210 -0
  158. reader_workbench/plugins/transform/__init__.py +0 -0
  159. reader_workbench/plugins/transform/_four_state_vector.py +204 -0
  160. reader_workbench/plugins/transform/_labeling.py +109 -0
  161. reader_workbench/plugins/transform/alias.py +70 -0
  162. reader_workbench/plugins/transform/assay_labels.py +62 -0
  163. reader_workbench/plugins/transform/blank.py +79 -0
  164. reader_workbench/plugins/transform/crosstalk_pairs.py +180 -0
  165. reader_workbench/plugins/transform/cytometry_gating.py +120 -0
  166. reader_workbench/plugins/transform/fold_change.py +79 -0
  167. reader_workbench/plugins/transform/four_state_event_window.py +93 -0
  168. reader_workbench/plugins/transform/four_state_vector.py +62 -0
  169. reader_workbench/plugins/transform/four_state_vector_collection.py +41 -0
  170. reader_workbench/plugins/transform/logic_symmetry.py +67 -0
  171. reader_workbench/plugins/transform/outlier_filter.py +60 -0
  172. reader_workbench/plugins/transform/overflow.py +197 -0
  173. reader_workbench/plugins/transform/ratio.py +237 -0
  174. reader_workbench/plugins/transform/sample_map.py +170 -0
  175. reader_workbench/plugins/transform/sample_metadata.py +94 -0
  176. reader_workbench/plugins/validator/__init__.py +1 -0
  177. reader_workbench/plugins/validator/to_tidy_plus_map.py +155 -0
  178. reader_workbench/protocols/__init__.py +80 -0
  179. reader_workbench/protocols/_builtins_plate_reader_growth.py +179 -0
  180. reader_workbench/protocols/_builtins_plate_reader_variants.py +274 -0
  181. reader_workbench/protocols/builtins.py +1656 -0
  182. reader_workbench/protocols/compiler.py +22 -0
  183. reader_workbench/protocols/compilers/__init__.py +1 -0
  184. reader_workbench/protocols/compilers/common.py +100 -0
  185. reader_workbench/protocols/compilers/cytometry.py +87 -0
  186. reader_workbench/protocols/compilers/generic.py +14 -0
  187. reader_workbench/protocols/compilers/logic.py +245 -0
  188. reader_workbench/protocols/compilers/plate_reader.py +937 -0
  189. reader_workbench/protocols/compilers/plate_reader_pipeline.py +197 -0
  190. reader_workbench/protocols/model.py +1486 -0
  191. reader_workbench/protocols/semantic_coverage.py +234 -0
  192. reader_workbench/runtime/__init__.py +12 -0
  193. reader_workbench/runtime/builtin.py +23 -0
  194. reader_workbench/runtime/model.py +42 -0
  195. reader_workbench/workbench/__init__.py +60 -0
  196. reader_workbench/workbench/assets/__init__.py +22 -0
  197. reader_workbench/workbench/assets/types.py +118 -0
  198. reader_workbench/workbench/audit/__init__.py +5 -0
  199. reader_workbench/workbench/audit/experiments.py +307 -0
  200. reader_workbench/workbench/audit/staging.py +187 -0
  201. reader_workbench/workbench/cli/__init__.py +51 -0
  202. reader_workbench/workbench/cli/_lazy.py +9 -0
  203. reader_workbench/workbench/cli/_records_view.py +150 -0
  204. reader_workbench/workbench/cli/_surface_execution.py +443 -0
  205. reader_workbench/workbench/cli/audit.py +95 -0
  206. reader_workbench/workbench/cli/automation.py +229 -0
  207. reader_workbench/workbench/cli/demo.py +46 -0
  208. reader_workbench/workbench/cli/dop.py +91 -0
  209. reader_workbench/workbench/cli/experiments.py +635 -0
  210. reader_workbench/workbench/cli/helpers.py +232 -0
  211. reader_workbench/workbench/cli/main.py +59 -0
  212. reader_workbench/workbench/cli/maintenance.py +82 -0
  213. reader_workbench/workbench/cli/notebooks.py +260 -0
  214. reader_workbench/workbench/cli/pagination.py +117 -0
  215. reader_workbench/workbench/cli/protocols.py +336 -0
  216. reader_workbench/workbench/cli/shared.py +309 -0
  217. reader_workbench/workbench/cli/surfaces.py +534 -0
  218. reader_workbench/workbench/cli/verification.py +128 -0
  219. reader_workbench/workbench/commands.py +10 -0
  220. reader_workbench/workbench/config/__init__.py +47 -0
  221. reader_workbench/workbench/config/identity.py +13 -0
  222. reader_workbench/workbench/config/load.py +405 -0
  223. reader_workbench/workbench/config/model.py +274 -0
  224. reader_workbench/workbench/context.py +26 -0
  225. reader_workbench/workbench/decl/__init__.py +31 -0
  226. reader_workbench/workbench/decl/build.py +190 -0
  227. reader_workbench/workbench/decl/model.py +81 -0
  228. reader_workbench/workbench/dop/__init__.py +12 -0
  229. reader_workbench/workbench/dop/builtins.py +261 -0
  230. reader_workbench/workbench/dop/model.py +209 -0
  231. reader_workbench/workbench/engine/__init__.py +42 -0
  232. reader_workbench/workbench/engine/_shared.py +76 -0
  233. reader_workbench/workbench/engine/contracts.py +283 -0
  234. reader_workbench/workbench/engine/execution.py +326 -0
  235. reader_workbench/workbench/engine/file_outputs.py +260 -0
  236. reader_workbench/workbench/engine/inputs.py +161 -0
  237. reader_workbench/workbench/engine/invocations.py +507 -0
  238. reader_workbench/workbench/engine/planning.py +72 -0
  239. reader_workbench/workbench/engine/runtime.py +464 -0
  240. reader_workbench/workbench/engine/setup.py +149 -0
  241. reader_workbench/workbench/engine/validation.py +684 -0
  242. reader_workbench/workbench/experiment/__init__.py +47 -0
  243. reader_workbench/workbench/experiment/model.py +381 -0
  244. reader_workbench/workbench/experiments.py +133 -0
  245. reader_workbench/workbench/graph/__init__.py +47 -0
  246. reader_workbench/workbench/graph/nodes.py +102 -0
  247. reader_workbench/workbench/graph/normalize.py +177 -0
  248. reader_workbench/workbench/graph/refs.py +148 -0
  249. reader_workbench/workbench/input_discovery.py +19 -0
  250. reader_workbench/workbench/inspection/__init__.py +3 -0
  251. reader_workbench/workbench/inspection/catalogs.py +128 -0
  252. reader_workbench/workbench/inspection/common.py +92 -0
  253. reader_workbench/workbench/inspection/dop.py +64 -0
  254. reader_workbench/workbench/inspection/experiments.py +449 -0
  255. reader_workbench/workbench/inspection/inventory.py +68 -0
  256. reader_workbench/workbench/inspection/protocols.py +368 -0
  257. reader_workbench/workbench/inspection/readiness.py +333 -0
  258. reader_workbench/workbench/inspection/reports.py +367 -0
  259. reader_workbench/workbench/inspection/results.py +166 -0
  260. reader_workbench/workbench/inspection/runtime.py +287 -0
  261. reader_workbench/workbench/inspection/semantics.py +192 -0
  262. reader_workbench/workbench/inspection/validation.py +30 -0
  263. reader_workbench/workbench/notebooks/__init__.py +17 -0
  264. reader_workbench/workbench/notebooks/_launch_registry.py +112 -0
  265. reader_workbench/workbench/notebooks/_launch_runtime.py +104 -0
  266. reader_workbench/workbench/notebooks/components/__init__.py +21 -0
  267. reader_workbench/workbench/notebooks/components/deliverables.py +403 -0
  268. reader_workbench/workbench/notebooks/components/overview.py +119 -0
  269. reader_workbench/workbench/notebooks/eda.marimo.py.txt +153 -0
  270. reader_workbench/workbench/notebooks/launch.py +274 -0
  271. reader_workbench/workbench/notebooks/presentation.py +136 -0
  272. reader_workbench/workbench/notebooks/scaffold.py +60 -0
  273. reader_workbench/workbench/ontology.py +78 -0
  274. reader_workbench/workbench/paths.py +44 -0
  275. reader_workbench/workbench/ports/__init__.py +31 -0
  276. reader_workbench/workbench/ports/model.py +168 -0
  277. reader_workbench/workbench/records/__init__.py +44 -0
  278. reader_workbench/workbench/records/epoch.py +329 -0
  279. reader_workbench/workbench/records/evidence.py +247 -0
  280. reader_workbench/workbench/records/identity.py +87 -0
  281. reader_workbench/workbench/records/locking.py +185 -0
  282. reader_workbench/workbench/records/model.py +711 -0
  283. reader_workbench/workbench/records/sources.py +73 -0
  284. reader_workbench/workbench/records/store.py +1022 -0
  285. reader_workbench/workbench/records/verification.py +998 -0
  286. reader_workbench/workbench/registry.py +333 -0
  287. reader_workbench/workbench/spec_overrides.py +215 -0
  288. reader_workbench-1.0.0.dist-info/METADATA +91 -0
  289. reader_workbench-1.0.0.dist-info/RECORD +293 -0
  290. reader_workbench-1.0.0.dist-info/WHEEL +5 -0
  291. reader_workbench-1.0.0.dist-info/entry_points.txt +2 -0
  292. reader_workbench-1.0.0.dist-info/licenses/LICENSE +21 -0
  293. reader_workbench-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,711 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import os
5
+ import stat
6
+ from collections import Counter
7
+ from dataclasses import dataclass, field
8
+ from io import BytesIO
9
+ from pathlib import Path
10
+ from typing import Any
11
+
12
+ import pandas as pd
13
+ import pyarrow.parquet as pq
14
+
15
+ from reader_workbench.errors import RecordError
16
+ from reader_workbench.workbench.ontology import WorkbenchProducerKind, WorkbenchRecordKind
17
+
18
+ from .evidence import ArtifactEvidence, RecordInputEvidence, SourceExperimentResolver
19
+ from .identity import BuildIdentity, digest_json, is_sha256_digest
20
+
21
+ RECORD_SCHEMA_VERSION = 6
22
+ _V6_BASE_FIELDS = {
23
+ "schema_version",
24
+ "record_id",
25
+ "kind",
26
+ "producer",
27
+ "created_at",
28
+ "inputs",
29
+ "config_digest",
30
+ }
31
+ _V6_DATAFRAME_FIELDS = _V6_BASE_FIELDS | {
32
+ "contract_id",
33
+ "path",
34
+ "content_digest",
35
+ "code_digest",
36
+ "producer_config_digest",
37
+ "build_identity",
38
+ "size_bytes",
39
+ }
40
+ _V6_FILE_BUNDLE_REQUIRED_FIELDS = _V6_BASE_FIELDS | {
41
+ "files",
42
+ "description",
43
+ "file_evidence",
44
+ "producer_config_digest",
45
+ "build_identity",
46
+ }
47
+ _V6_FILE_BUNDLE_FIELDS = _V6_FILE_BUNDLE_REQUIRED_FIELDS | {"path_descriptions"}
48
+ _SOURCE_RECIPE_FIELDS = {"recipe", "with"}
49
+
50
+
51
+ @dataclass(frozen=True)
52
+ class RecordRecipeSource:
53
+ recipe: str
54
+ with_: dict[str, Any] = field(default_factory=dict)
55
+
56
+ def to_dict(self) -> dict[str, Any]:
57
+ return {"recipe": self.recipe, "with": dict(self.with_ or {})}
58
+
59
+
60
+ @dataclass(frozen=True)
61
+ class RecordProducer:
62
+ kind: WorkbenchProducerKind
63
+ id: str
64
+ plugin: str | None = None
65
+ source_recipe: RecordRecipeSource | None = None
66
+
67
+ def __post_init__(self) -> None:
68
+ if self.kind not in {"pipeline", "plot", "export"}:
69
+ raise RecordError(f"record producers must use pipeline, plot, or export kind; got {self.kind!r}")
70
+ if not isinstance(self.plugin, str) or not self.plugin:
71
+ raise RecordError("pipeline/plot/export producers must include a non-empty plugin id")
72
+
73
+ def to_dict(self) -> dict[str, Any]:
74
+ payload: dict[str, Any] = {"kind": self.kind, "id": self.id, "plugin": self.plugin or ""}
75
+ if self.source_recipe is not None:
76
+ payload["source_recipe"] = self.source_recipe.to_dict()
77
+ return payload
78
+
79
+
80
+ @dataclass(frozen=True)
81
+ class WorkbenchRecord:
82
+ record_id: str
83
+ kind: WorkbenchRecordKind
84
+ producer: RecordProducer
85
+ created_at: str
86
+ inputs: tuple[RecordInputEvidence, ...]
87
+ config_digest: str
88
+
89
+ def __post_init__(self) -> None:
90
+ inputs = tuple(self.inputs)
91
+ if any(not isinstance(item, RecordInputEvidence) for item in inputs):
92
+ raise RecordError("record inputs must contain immutable RecordInputEvidence values")
93
+ object.__setattr__(self, "inputs", inputs)
94
+
95
+
96
+ @dataclass(frozen=True)
97
+ class DataFrameArtifactRecord(WorkbenchRecord):
98
+ contract_id: str
99
+ path: Path
100
+ content_digest: str
101
+ code_digest: str = ""
102
+ producer_config_digest: str = ""
103
+ build_identity: BuildIdentity | None = None
104
+ size_bytes: int | None = None
105
+ schema_version: int = RECORD_SCHEMA_VERSION
106
+
107
+ def __post_init__(self) -> None:
108
+ super().__post_init__()
109
+ if self.kind != "dataframe_artifact":
110
+ raise RecordError(f"DataFrameArtifactRecord must use kind 'dataframe_artifact', got {self.kind!r}")
111
+ if self.schema_version != RECORD_SCHEMA_VERSION:
112
+ raise RecordError(
113
+ f"DataFrameArtifactRecord schema_version must be {RECORD_SCHEMA_VERSION} (got {self.schema_version!r})"
114
+ )
115
+ if not self.producer_config_digest:
116
+ raise RecordError("schema-v6 dataframe records require producer_config_digest")
117
+ if self.build_identity is None:
118
+ raise RecordError("schema-v6 dataframe records require build_identity")
119
+ if not isinstance(self.size_bytes, int) or self.size_bytes < 0:
120
+ raise RecordError("schema-v6 dataframe records require non-negative size_bytes")
121
+ if not is_sha256_digest(self.content_digest):
122
+ raise RecordError("schema-v6 dataframe records require a sha256 content_digest")
123
+ if not is_sha256_digest(self.code_digest):
124
+ raise RecordError("schema-v6 dataframe records require a sha256 code_digest")
125
+
126
+ def load_dataframe(self, *, row_limit: int | None = None) -> pd.DataFrame:
127
+ if row_limit is not None and (isinstance(row_limit, bool) or not isinstance(row_limit, int) or row_limit < 1):
128
+ raise RecordError("row_limit must be a positive integer")
129
+ if self.path.suffix.lower() == ".parquet":
130
+ if row_limit is not None:
131
+ return _read_verified_parquet_rows(
132
+ self.path,
133
+ record_id=self.record_id,
134
+ expected_size=self.size_bytes or 0,
135
+ expected_digest=self.content_digest,
136
+ row_limit=row_limit,
137
+ )
138
+ content = _read_verified_artifact_bytes(
139
+ self.path,
140
+ record_id=self.record_id,
141
+ expected_size=self.size_bytes or 0,
142
+ expected_digest=self.content_digest,
143
+ )
144
+ return pd.read_parquet(BytesIO(content))
145
+ raise RecordError(f"Record {self.record_id} is not a parquet dataframe: {self.path}")
146
+
147
+ def verify_content_digest(self) -> None:
148
+ _verify_artifact_file(
149
+ self.path,
150
+ record_id=self.record_id,
151
+ expected_size=self.size_bytes or 0,
152
+ expected_digest=self.content_digest,
153
+ )
154
+
155
+
156
+ def _verify_artifact_file(
157
+ path: Path,
158
+ *,
159
+ record_id: str,
160
+ expected_size: int,
161
+ expected_digest: str,
162
+ ) -> None:
163
+ try:
164
+ if not path.exists():
165
+ raise RecordError(f"Record {record_id} artifact is missing: {path}")
166
+ if not path.is_file():
167
+ raise RecordError(f"Record {record_id} artifact is not a regular file: {path}")
168
+ actual_size = path.stat().st_size
169
+ except OSError as exc:
170
+ raise RecordError(f"Record {record_id} artifact could not be read: {path}: {exc}") from exc
171
+ if actual_size != expected_size:
172
+ raise RecordError(
173
+ f"Record {record_id} content digest mismatch for {path}: "
174
+ f"artifact size changed from {expected_size} to {actual_size} bytes"
175
+ )
176
+ try:
177
+ actual_digest = sha256_file(path)
178
+ except OSError as exc:
179
+ raise RecordError(f"Record {record_id} artifact could not be read: {path}: {exc}") from exc
180
+ if actual_digest != expected_digest:
181
+ raise RecordError(
182
+ f"Record {record_id} content digest mismatch for {path}: expected {expected_digest}, got {actual_digest}"
183
+ )
184
+
185
+
186
+ def _read_verified_artifact_bytes(
187
+ path: Path,
188
+ *,
189
+ record_id: str,
190
+ expected_size: int,
191
+ expected_digest: str,
192
+ ) -> bytes:
193
+ """Read, verify, and return one immutable buffer from one open descriptor."""
194
+
195
+ try:
196
+ with path.open("rb") as artifact:
197
+ artifact_stat = os.fstat(artifact.fileno())
198
+ if not stat.S_ISREG(artifact_stat.st_mode):
199
+ raise RecordError(f"Record {record_id} artifact is not a regular file: {path}")
200
+ content = artifact.read()
201
+ except FileNotFoundError as exc:
202
+ raise RecordError(f"Record {record_id} artifact is missing: {path}") from exc
203
+ except IsADirectoryError as exc:
204
+ raise RecordError(f"Record {record_id} artifact is not a regular file: {path}") from exc
205
+ except RecordError:
206
+ raise
207
+ except OSError as exc:
208
+ raise RecordError(f"Record {record_id} artifact could not be read: {path}: {exc}") from exc
209
+ actual_size = len(content)
210
+ if actual_size != expected_size:
211
+ raise RecordError(
212
+ f"Record {record_id} content digest mismatch for {path}: "
213
+ f"artifact size changed from {expected_size} to {actual_size} bytes"
214
+ )
215
+ actual_digest = "sha256:" + hashlib.sha256(content).hexdigest()
216
+ if actual_digest != expected_digest:
217
+ raise RecordError(
218
+ f"Record {record_id} content digest mismatch for {path}: expected {expected_digest}, got {actual_digest}"
219
+ )
220
+ return content
221
+
222
+
223
+ def _read_verified_parquet_rows(
224
+ path: Path,
225
+ *,
226
+ record_id: str,
227
+ expected_size: int,
228
+ expected_digest: str,
229
+ row_limit: int,
230
+ ) -> pd.DataFrame:
231
+ """Verify one open parquet descriptor and materialize at most ``row_limit`` rows."""
232
+
233
+ try:
234
+ with path.open("rb") as artifact:
235
+ artifact_stat = os.fstat(artifact.fileno())
236
+ if not stat.S_ISREG(artifact_stat.st_mode):
237
+ raise RecordError(f"Record {record_id} artifact is not a regular file: {path}")
238
+ if artifact_stat.st_size != expected_size:
239
+ raise RecordError(
240
+ f"Record {record_id} content digest mismatch for {path}: "
241
+ f"artifact size changed from {expected_size} to {artifact_stat.st_size} bytes"
242
+ )
243
+ digest = hashlib.sha256()
244
+ for chunk in iter(lambda: artifact.read(1024 * 1024), b""):
245
+ digest.update(chunk)
246
+ actual_digest = "sha256:" + digest.hexdigest()
247
+ if actual_digest != expected_digest:
248
+ raise RecordError(
249
+ f"Record {record_id} content digest mismatch for {path}: "
250
+ f"expected {expected_digest}, got {actual_digest}"
251
+ )
252
+ artifact.seek(0)
253
+ parquet_file = pq.ParquetFile(artifact)
254
+ first_batch = next(parquet_file.iter_batches(batch_size=row_limit), None)
255
+ dataframe = (
256
+ parquet_file.schema_arrow.empty_table().to_pandas() if first_batch is None else first_batch.to_pandas()
257
+ )
258
+ final_stat = os.fstat(artifact.fileno())
259
+ if (
260
+ final_stat.st_dev,
261
+ final_stat.st_ino,
262
+ final_stat.st_size,
263
+ final_stat.st_mtime_ns,
264
+ ) != (
265
+ artifact_stat.st_dev,
266
+ artifact_stat.st_ino,
267
+ artifact_stat.st_size,
268
+ artifact_stat.st_mtime_ns,
269
+ ):
270
+ raise RecordError(f"Record {record_id} artifact changed while it was being read: {path}")
271
+ return dataframe
272
+ except FileNotFoundError as exc:
273
+ raise RecordError(f"Record {record_id} artifact is missing: {path}") from exc
274
+ except IsADirectoryError as exc:
275
+ raise RecordError(f"Record {record_id} artifact is not a regular file: {path}") from exc
276
+ except RecordError:
277
+ raise
278
+ except OSError as exc:
279
+ raise RecordError(f"Record {record_id} artifact could not be read: {path}: {exc}") from exc
280
+
281
+
282
+ def sha256_file(path: Path, *, chunk_size: int = 1024 * 1024) -> str:
283
+ digest = hashlib.sha256()
284
+ with path.open("rb") as artifact:
285
+ for chunk in iter(lambda: artifact.read(chunk_size), b""):
286
+ digest.update(chunk)
287
+ return "sha256:" + digest.hexdigest()
288
+
289
+
290
+ def _normalized_description(value: Any, *, where: str) -> str:
291
+ if not isinstance(value, str) or not value.strip():
292
+ raise RecordError(f"{where} must be a non-empty string")
293
+ normalized = value.strip()
294
+ if "\n" in normalized or "\r" in normalized:
295
+ raise RecordError(f"{where} must be a single line")
296
+ return normalized
297
+
298
+
299
+ @dataclass(frozen=True)
300
+ class PathDescription:
301
+ path: Path
302
+ description: str
303
+
304
+ def __post_init__(self) -> None:
305
+ try:
306
+ path = Path(self.path)
307
+ except (TypeError, ValueError) as exc:
308
+ raise RecordError("PathDescription path must be path-like") from exc
309
+ if not str(path).strip() or str(path) == ".":
310
+ raise RecordError("PathDescription path must be a non-empty path")
311
+ object.__setattr__(self, "path", path)
312
+ object.__setattr__(
313
+ self,
314
+ "description",
315
+ _normalized_description(self.description, where="PathDescription description"),
316
+ )
317
+
318
+
319
+ def normalize_file_bundle_metadata(
320
+ *,
321
+ producer_kind: WorkbenchProducerKind,
322
+ files: tuple[Path, ...],
323
+ description: str | None,
324
+ path_descriptions: tuple[PathDescription, ...],
325
+ ) -> tuple[str, tuple[PathDescription, ...]]:
326
+ """Validate and normalize the non-filesystem portion of a file-bundle contract."""
327
+
328
+ if not files:
329
+ raise RecordError("FileBundleRecord must contain at least one file")
330
+ normalized_description = _normalized_description(description, where="FileBundleRecord description")
331
+ try:
332
+ descriptions = tuple(path_descriptions)
333
+ except TypeError as exc:
334
+ raise RecordError("FileBundleRecord path description entries must be PathDescription values") from exc
335
+ if any(not isinstance(item, PathDescription) for item in descriptions):
336
+ raise RecordError("FileBundleRecord path description entries must be PathDescription values")
337
+ if producer_kind == "plot" and not descriptions:
338
+ raise RecordError("plot file bundles must describe every file")
339
+
340
+ described_paths = [item.path for item in descriptions]
341
+ duplicate_paths = sorted(
342
+ (path for path, count in Counter(described_paths).items() if count > 1),
343
+ key=str,
344
+ )
345
+ if duplicate_paths:
346
+ raise RecordError(
347
+ "FileBundleRecord has duplicate path descriptions: " + ", ".join(str(path) for path in duplicate_paths)
348
+ )
349
+ file_paths = set(files)
350
+ unmatched = sorted(set(described_paths) - file_paths, key=str)
351
+ if unmatched:
352
+ raise RecordError(
353
+ "FileBundleRecord has unmatched path descriptions: " + ", ".join(str(path) for path in unmatched)
354
+ )
355
+ if described_paths:
356
+ missing = sorted(file_paths - set(described_paths), key=str)
357
+ if missing:
358
+ raise RecordError(
359
+ "FileBundleRecord has missing path descriptions: " + ", ".join(str(path) for path in missing)
360
+ )
361
+ descriptions_by_path = {item.path: item for item in descriptions}
362
+ descriptions = tuple(descriptions_by_path[path] for path in files)
363
+ return normalized_description, descriptions
364
+
365
+
366
+ @dataclass(frozen=True)
367
+ class FileBundleRecord(WorkbenchRecord):
368
+ files: tuple[Path, ...]
369
+ description: str | None
370
+ path_descriptions: tuple[PathDescription, ...] = ()
371
+ file_evidence: tuple[ArtifactEvidence, ...] = ()
372
+ producer_config_digest: str = ""
373
+ build_identity: BuildIdentity | None = None
374
+ schema_version: int = RECORD_SCHEMA_VERSION
375
+
376
+ def __post_init__(self) -> None:
377
+ super().__post_init__()
378
+ if self.kind != "file_bundle":
379
+ raise RecordError(f"FileBundleRecord must use kind 'file_bundle', got {self.kind!r}")
380
+ if self.schema_version != RECORD_SCHEMA_VERSION:
381
+ raise RecordError(
382
+ f"FileBundleRecord schema_version must be {RECORD_SCHEMA_VERSION} (got {self.schema_version!r})"
383
+ )
384
+ description, path_descriptions = normalize_file_bundle_metadata(
385
+ producer_kind=self.producer.kind,
386
+ files=self.files,
387
+ description=self.description,
388
+ path_descriptions=self.path_descriptions,
389
+ )
390
+ object.__setattr__(self, "description", description)
391
+ object.__setattr__(self, "path_descriptions", path_descriptions)
392
+ if not self.producer_config_digest:
393
+ raise RecordError("schema-v6 file-bundle records require producer_config_digest")
394
+ if self.build_identity is None:
395
+ raise RecordError("schema-v6 file-bundle records require build_identity")
396
+ evidence = tuple(self.file_evidence)
397
+ if len(evidence) != len(self.files):
398
+ raise RecordError("schema-v6 file-bundle records must include evidence for every file")
399
+ evidence_paths = [item.relative_path for item in evidence]
400
+ if len(set(evidence_paths)) != len(evidence_paths):
401
+ raise RecordError("schema-v6 file-bundle records contain duplicate evidence paths")
402
+ object.__setattr__(self, "file_evidence", evidence)
403
+
404
+ def description_for(self, path: Path) -> str | None:
405
+ requested = Path(path)
406
+ for item in self.path_descriptions:
407
+ if item.path == requested:
408
+ return item.description
409
+ return None
410
+
411
+
412
+ def record_paths(record: DataFrameArtifactRecord | FileBundleRecord) -> tuple[Path, ...]:
413
+ if isinstance(record, DataFrameArtifactRecord):
414
+ return (record.path,)
415
+ return record.files
416
+
417
+
418
+ def _path_within_outputs(path: Path, *, outputs_dir: Path) -> tuple[Path, Path]:
419
+ try:
420
+ outputs_root = outputs_dir.resolve(strict=False)
421
+ candidate = path if path.is_absolute() else outputs_root / path
422
+ resolved = candidate.resolve(strict=False)
423
+ relative = resolved.relative_to(outputs_root)
424
+ except ValueError as exc:
425
+ raise RecordError(f"record path {path!s} must resolve within the outputs directory") from exc
426
+ except (OSError, RuntimeError) as exc:
427
+ raise RecordError(f"record path {path!s} could not be resolved safely: {exc}") from exc
428
+ if relative == Path("."):
429
+ raise RecordError("record paths must identify files below the outputs directory")
430
+ return resolved, relative
431
+
432
+
433
+ def verify_record_artifact_integrity(
434
+ record: DataFrameArtifactRecord | FileBundleRecord,
435
+ *,
436
+ outputs_dir: Path,
437
+ ) -> None:
438
+ """Verify every confined artifact bound to one exact record revision."""
439
+
440
+ if isinstance(record, DataFrameArtifactRecord):
441
+ resolved, _relative = _path_within_outputs(record.path, outputs_dir=outputs_dir)
442
+ _verify_artifact_file(
443
+ resolved,
444
+ record_id=record.record_id,
445
+ expected_size=record.size_bytes or 0,
446
+ expected_digest=record.content_digest,
447
+ )
448
+ return
449
+
450
+ evidence_by_path = {item.relative_path: item for item in record.file_evidence}
451
+ files_by_path: dict[Path, Path] = {}
452
+ for path in record.files:
453
+ resolved, relative = _path_within_outputs(path, outputs_dir=outputs_dir)
454
+ files_by_path[relative] = resolved
455
+ if set(files_by_path) != set(evidence_by_path):
456
+ missing_evidence = sorted(set(files_by_path) - set(evidence_by_path), key=str)
457
+ orphan_evidence = sorted(set(evidence_by_path) - set(files_by_path), key=str)
458
+ details: list[str] = []
459
+ if missing_evidence:
460
+ details.append("missing evidence for " + ", ".join(map(str, missing_evidence)))
461
+ if orphan_evidence:
462
+ details.append("orphan evidence for " + ", ".join(map(str, orphan_evidence)))
463
+ raise RecordError(f"Record {record.record_id} artifact evidence paths do not match: {'; '.join(details)}")
464
+ for relative, path in files_by_path.items():
465
+ evidence = evidence_by_path[relative]
466
+ _verify_artifact_file(
467
+ path,
468
+ record_id=record.record_id,
469
+ expected_size=evidence.size_bytes,
470
+ expected_digest=evidence.content_digest,
471
+ )
472
+
473
+
474
+ def _decode_path(raw: Any, *, outputs_dir: Path) -> Path:
475
+ if not isinstance(raw, str) or not raw.strip():
476
+ raise RecordError("record path entries must be non-empty strings")
477
+ path = Path(raw)
478
+ if path.is_absolute():
479
+ raise RecordError("record path entries must be relative to the outputs directory")
480
+ resolved, _relative = _path_within_outputs(path, outputs_dir=outputs_dir)
481
+ return resolved
482
+
483
+
484
+ def _encode_path(path: Path, *, outputs_dir: Path) -> str:
485
+ _resolved, relative = _path_within_outputs(Path(path), outputs_dir=outputs_dir)
486
+ return str(relative)
487
+
488
+
489
+ def record_to_dict(record: DataFrameArtifactRecord | FileBundleRecord, *, outputs_dir: Path) -> dict[str, Any]:
490
+ base = {
491
+ "schema_version": record.schema_version,
492
+ "record_id": record.record_id,
493
+ "kind": record.kind,
494
+ "producer": record.producer.to_dict(),
495
+ "created_at": record.created_at,
496
+ "inputs": [item.to_dict() for item in record.inputs],
497
+ "config_digest": record.config_digest,
498
+ }
499
+ if isinstance(record, DataFrameArtifactRecord):
500
+ base.update(
501
+ {
502
+ "contract_id": record.contract_id,
503
+ "path": _encode_path(record.path, outputs_dir=outputs_dir),
504
+ "content_digest": record.content_digest,
505
+ "code_digest": record.code_digest,
506
+ }
507
+ )
508
+ base.update(
509
+ {
510
+ "producer_config_digest": record.producer_config_digest,
511
+ "build_identity": record.build_identity.to_dict() if record.build_identity else None,
512
+ "size_bytes": record.size_bytes,
513
+ }
514
+ )
515
+ return base
516
+ base["files"] = [_encode_path(path, outputs_dir=outputs_dir) for path in record.files]
517
+ base["description"] = record.description
518
+ if record.path_descriptions:
519
+ base["path_descriptions"] = [
520
+ {
521
+ "path": _encode_path(item.path, outputs_dir=outputs_dir),
522
+ "description": item.description,
523
+ }
524
+ for item in record.path_descriptions
525
+ ]
526
+ base.update(
527
+ {
528
+ "producer_config_digest": record.producer_config_digest,
529
+ "build_identity": record.build_identity.to_dict() if record.build_identity else None,
530
+ "file_evidence": [item.to_dict() for item in record.file_evidence],
531
+ }
532
+ )
533
+ return base
534
+
535
+
536
+ def record_revision_digest(record: DataFrameArtifactRecord | FileBundleRecord, *, outputs_dir: Path) -> str:
537
+ """Return the stable identity of one exact persisted record revision."""
538
+ return digest_json(record_to_dict(record, outputs_dir=outputs_dir))
539
+
540
+
541
+ def record_from_dict(
542
+ payload: dict[str, Any],
543
+ *,
544
+ outputs_dir: Path,
545
+ experiment_root: Path | None = None,
546
+ source_experiment_resolver: SourceExperimentResolver | None = None,
547
+ ) -> DataFrameArtifactRecord | FileBundleRecord:
548
+ if not isinstance(payload, dict):
549
+ raise RecordError("record payload must be a JSON object")
550
+ schema_version = payload.get("schema_version")
551
+ record_id = payload.get("record_id")
552
+ kind = payload.get("kind")
553
+ producer_payload = payload.get("producer")
554
+ created_at = payload.get("created_at")
555
+ inputs = payload.get("inputs")
556
+ config_digest = payload.get("config_digest")
557
+ if not isinstance(record_id, str) or not record_id:
558
+ raise RecordError("record payload must include non-empty string 'record_id'")
559
+ if kind not in {"dataframe_artifact", "file_bundle"}:
560
+ raise RecordError(f"record {record_id!r} has unknown kind {kind!r}")
561
+ if schema_version != RECORD_SCHEMA_VERSION:
562
+ raise RecordError(
563
+ f"record payload schema_version must be {RECORD_SCHEMA_VERSION} (got {schema_version!r}); "
564
+ "regenerate the catalog with `reader run <config|dir|index> --reset-records`"
565
+ )
566
+ allowed = _V6_DATAFRAME_FIELDS if kind == "dataframe_artifact" else _V6_FILE_BUNDLE_FIELDS
567
+ required = _V6_DATAFRAME_FIELDS if kind == "dataframe_artifact" else _V6_FILE_BUNDLE_REQUIRED_FIELDS
568
+ unknown = sorted(set(payload) - allowed)
569
+ missing = sorted(required - set(payload))
570
+ if unknown or missing:
571
+ details = []
572
+ if unknown:
573
+ details.append("unknown=" + ", ".join(unknown))
574
+ if missing:
575
+ details.append("missing=" + ", ".join(missing))
576
+ raise RecordError(f"schema-v6 record payload has unknown or missing fields: {'; '.join(details)}")
577
+ if not isinstance(producer_payload, dict):
578
+ raise RecordError(f"record {record_id!r} must include producer metadata")
579
+ producer_kind = producer_payload.get("kind")
580
+ producer_id = producer_payload.get("id")
581
+ producer_plugin = producer_payload.get("plugin")
582
+ source_recipe_payload = producer_payload.get("source_recipe")
583
+ unknown_producer_fields = sorted(set(producer_payload) - {"kind", "id", "plugin", "source_recipe"})
584
+ if unknown_producer_fields:
585
+ raise RecordError(f"record {record_id!r} producer has unknown fields: {', '.join(unknown_producer_fields)}")
586
+ if producer_kind not in {"pipeline", "plot", "export"}:
587
+ raise RecordError(f"record {record_id!r} has invalid producer kind {producer_kind!r}")
588
+ if not isinstance(producer_id, str) or not producer_id:
589
+ raise RecordError(f"record {record_id!r} must include producer.id")
590
+ if not isinstance(producer_plugin, str) or not producer_plugin:
591
+ raise RecordError(f"record {record_id!r} must include producer.plugin")
592
+ source_recipe = None
593
+ if source_recipe_payload is not None:
594
+ if not isinstance(source_recipe_payload, dict):
595
+ raise RecordError(f"record {record_id!r} must include producer.source_recipe as an object")
596
+ unknown_source_recipe_fields = sorted(set(source_recipe_payload) - _SOURCE_RECIPE_FIELDS)
597
+ missing_source_recipe_fields = sorted(_SOURCE_RECIPE_FIELDS - set(source_recipe_payload))
598
+ if unknown_source_recipe_fields or missing_source_recipe_fields:
599
+ details = []
600
+ if unknown_source_recipe_fields:
601
+ details.append("unknown=" + ", ".join(unknown_source_recipe_fields))
602
+ if missing_source_recipe_fields:
603
+ details.append("missing=" + ", ".join(missing_source_recipe_fields))
604
+ raise RecordError(
605
+ f"record {record_id!r} producer.source_recipe has unknown or missing fields: {'; '.join(details)}"
606
+ )
607
+ recipe_name = source_recipe_payload["recipe"]
608
+ with_block = source_recipe_payload["with"]
609
+ if not isinstance(recipe_name, str) or not recipe_name:
610
+ raise RecordError(f"record {record_id!r} must include producer.source_recipe.recipe")
611
+ if not isinstance(with_block, dict):
612
+ raise RecordError(f"record {record_id!r} producer.source_recipe.with must be a mapping")
613
+ source_recipe = RecordRecipeSource(recipe=recipe_name, with_=dict(with_block))
614
+ if not isinstance(created_at, str) or not created_at:
615
+ raise RecordError(f"record {record_id!r} must include created_at")
616
+ if not isinstance(inputs, list) or any(not isinstance(item, dict) for item in inputs):
617
+ raise RecordError(f"record {record_id!r} must include structured inputs")
618
+ if not isinstance(config_digest, str) or not config_digest:
619
+ raise RecordError(f"record {record_id!r} must include config_digest")
620
+ producer = RecordProducer(
621
+ kind=producer_kind,
622
+ id=producer_id,
623
+ plugin=producer_plugin if isinstance(producer_plugin, str) else None,
624
+ source_recipe=source_recipe,
625
+ )
626
+ try:
627
+ evidence_root = (experiment_root or outputs_dir.parent).resolve(strict=False)
628
+ parsed_inputs = tuple(
629
+ RecordInputEvidence.from_dict(
630
+ item,
631
+ experiment_root=evidence_root,
632
+ source_experiment_resolver=source_experiment_resolver,
633
+ )
634
+ for item in inputs
635
+ )
636
+ except (TypeError, ValueError, RecordError) as exc:
637
+ raise RecordError(f"record {record_id!r} has invalid provenance input: {exc}") from exc
638
+ if kind == "dataframe_artifact":
639
+ contract_id = payload.get("contract_id")
640
+ path = payload.get("path")
641
+ content_digest = payload.get("content_digest")
642
+ code_digest = payload.get("code_digest", "")
643
+ producer_config_digest = payload.get("producer_config_digest", "")
644
+ build_identity = BuildIdentity.from_dict(payload.get("build_identity"))
645
+ size_bytes = payload.get("size_bytes")
646
+ if not isinstance(contract_id, str) or not contract_id:
647
+ raise RecordError(f"record {record_id!r} must include contract_id")
648
+ if not isinstance(content_digest, str) or not content_digest:
649
+ raise RecordError(f"record {record_id!r} must include content_digest")
650
+ if not isinstance(code_digest, str):
651
+ raise RecordError(f"record {record_id!r} must include string code_digest")
652
+ return DataFrameArtifactRecord(
653
+ record_id=record_id,
654
+ kind="dataframe_artifact",
655
+ producer=producer,
656
+ created_at=created_at,
657
+ inputs=parsed_inputs,
658
+ config_digest=config_digest,
659
+ contract_id=contract_id,
660
+ path=_decode_path(path, outputs_dir=outputs_dir),
661
+ content_digest=content_digest,
662
+ code_digest=code_digest,
663
+ producer_config_digest=producer_config_digest,
664
+ build_identity=build_identity,
665
+ size_bytes=size_bytes,
666
+ schema_version=schema_version,
667
+ )
668
+ files = payload.get("files")
669
+ if not isinstance(files, list) or any(not isinstance(item, str) or not item for item in files):
670
+ raise RecordError(f"record {record_id!r} must include non-empty string file paths")
671
+ description = payload.get("description")
672
+ if not isinstance(description, str) or not description.strip():
673
+ raise RecordError(f"file-bundle record {record_id!r} must include a non-empty description")
674
+ path_descriptions_payload = payload.get("path_descriptions", [])
675
+ if not isinstance(path_descriptions_payload, list):
676
+ raise RecordError(f"record {record_id!r} path_descriptions must be a list when provided")
677
+ path_descriptions: list[PathDescription] = []
678
+ for item in path_descriptions_payload:
679
+ if not isinstance(item, dict) or set(item) != {"path", "description"}:
680
+ raise RecordError(f"record {record_id!r} path_descriptions entries must contain only path and description")
681
+ item_path = item.get("path")
682
+ item_description = item.get("description")
683
+ if not isinstance(item_description, str):
684
+ raise RecordError(f"record {record_id!r} path_descriptions entries must include string descriptions")
685
+ path_descriptions.append(
686
+ PathDescription(
687
+ path=_decode_path(item_path, outputs_dir=outputs_dir),
688
+ description=item_description,
689
+ )
690
+ )
691
+ file_evidence_payload = payload.get("file_evidence")
692
+ if not isinstance(file_evidence_payload, list):
693
+ raise RecordError(f"record {record_id!r} file_evidence must be a list")
694
+ file_evidence = tuple(ArtifactEvidence.from_dict(item) for item in file_evidence_payload)
695
+ producer_config_digest = payload.get("producer_config_digest", "")
696
+ build_identity = BuildIdentity.from_dict(payload.get("build_identity"))
697
+ return FileBundleRecord(
698
+ record_id=record_id,
699
+ kind="file_bundle",
700
+ producer=producer,
701
+ created_at=created_at,
702
+ inputs=parsed_inputs,
703
+ config_digest=config_digest,
704
+ files=tuple(_decode_path(path, outputs_dir=outputs_dir) for path in files),
705
+ description=description,
706
+ schema_version=schema_version,
707
+ path_descriptions=tuple(path_descriptions),
708
+ file_evidence=file_evidence,
709
+ producer_config_digest=producer_config_digest,
710
+ build_identity=build_identity,
711
+ )