reader-workbench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (293) hide show
  1. reader_workbench/__init__.py +22 -0
  2. reader_workbench/__main__.py +4 -0
  3. reader_workbench/_version.py +17 -0
  4. reader_workbench/api/__init__.py +74 -0
  5. reader_workbench/api/_record_reads.py +75 -0
  6. reader_workbench/api/artifacts.py +79 -0
  7. reader_workbench/api/facade.py +538 -0
  8. reader_workbench/api/models.py +285 -0
  9. reader_workbench/api/notebooks.py +63 -0
  10. reader_workbench/contracts/__init__.py +18 -0
  11. reader_workbench/contracts/builtins/__init__.py +36 -0
  12. reader_workbench/contracts/builtins/cytometry.py +140 -0
  13. reader_workbench/contracts/builtins/four_state_event_window.py +214 -0
  14. reader_workbench/contracts/builtins/generic.py +21 -0
  15. reader_workbench/contracts/builtins/logic.py +149 -0
  16. reader_workbench/contracts/builtins/plate_reader.py +47 -0
  17. reader_workbench/contracts/catalog.py +257 -0
  18. reader_workbench/contracts/model.py +109 -0
  19. reader_workbench/domains/__init__.py +1 -0
  20. reader_workbench/domains/cytometry/__init__.py +3 -0
  21. reader_workbench/domains/cytometry/analysis/__init__.py +29 -0
  22. reader_workbench/domains/cytometry/analysis/events.py +182 -0
  23. reader_workbench/domains/cytometry/analysis/gating.py +175 -0
  24. reader_workbench/domains/cytometry/analysis/workflow.py +248 -0
  25. reader_workbench/domains/cytometry/io/__init__.py +3 -0
  26. reader_workbench/domains/cytometry/io/fcs.py +135 -0
  27. reader_workbench/domains/cytometry/plots/__init__.py +5 -0
  28. reader_workbench/domains/cytometry/plots/diagnostic.py +155 -0
  29. reader_workbench/domains/logic/__init__.py +3 -0
  30. reader_workbench/domains/logic/crosstalk/__init__.py +3 -0
  31. reader_workbench/domains/logic/crosstalk/pairs.py +661 -0
  32. reader_workbench/domains/logic/four_state_vector/__init__.py +7 -0
  33. reader_workbench/domains/logic/four_state_vector/builder.py +321 -0
  34. reader_workbench/domains/logic/four_state_vector/collection/__init__.py +20 -0
  35. reader_workbench/domains/logic/four_state_vector/collection/checks.py +49 -0
  36. reader_workbench/domains/logic/four_state_vector/collection/constants.py +29 -0
  37. reader_workbench/domains/logic/four_state_vector/collection/model.py +19 -0
  38. reader_workbench/domains/logic/four_state_vector/collection/render.py +383 -0
  39. reader_workbench/domains/logic/four_state_vector/collection/sources.py +185 -0
  40. reader_workbench/domains/logic/four_state_vector/config.py +214 -0
  41. reader_workbench/domains/logic/four_state_vector/diagnostic.py +361 -0
  42. reader_workbench/domains/logic/four_state_vector/heatmap.py +86 -0
  43. reader_workbench/domains/logic/four_state_vector/math.py +191 -0
  44. reader_workbench/domains/logic/four_state_vector/reference.py +85 -0
  45. reader_workbench/domains/logic/four_state_vector/selection.py +228 -0
  46. reader_workbench/domains/logic/four_state_vector/treatment_semantics.py +51 -0
  47. reader_workbench/domains/logic/four_state_vector/validation.py +19 -0
  48. reader_workbench/domains/logic/logic_symmetry/__init__.py +3 -0
  49. reader_workbench/domains/logic/logic_symmetry/encodings.py +93 -0
  50. reader_workbench/domains/logic/logic_symmetry/extract_corners.py +192 -0
  51. reader_workbench/domains/logic/logic_symmetry/main.py +236 -0
  52. reader_workbench/domains/logic/logic_symmetry/metrics.py +96 -0
  53. reader_workbench/domains/logic/logic_symmetry/overlay.py +129 -0
  54. reader_workbench/domains/logic/logic_symmetry/prep.py +138 -0
  55. reader_workbench/domains/logic/logic_symmetry/render.py +356 -0
  56. reader_workbench/domains/logic/treatment_columns.py +42 -0
  57. reader_workbench/domains/plate_reader/__init__.py +1 -0
  58. reader_workbench/domains/plate_reader/analysis/__init__.py +14 -0
  59. reader_workbench/domains/plate_reader/analysis/fold_change.py +474 -0
  60. reader_workbench/domains/plate_reader/analysis/four_state_event_window/__init__.py +21 -0
  61. reader_workbench/domains/plate_reader/analysis/four_state_event_window/aggregation.py +191 -0
  62. reader_workbench/domains/plate_reader/analysis/four_state_event_window/contract_fields.py +50 -0
  63. reader_workbench/domains/plate_reader/analysis/four_state_event_window/contracts.py +320 -0
  64. reader_workbench/domains/plate_reader/analysis/four_state_event_window/design_dispositions.py +54 -0
  65. reader_workbench/domains/plate_reader/analysis/four_state_event_window/disposition_records.py +140 -0
  66. reader_workbench/domains/plate_reader/analysis/four_state_event_window/event_sensitivity.py +27 -0
  67. reader_workbench/domains/plate_reader/analysis/four_state_event_window/materialize.py +274 -0
  68. reader_workbench/domains/plate_reader/analysis/four_state_event_window/observation_resampling.py +97 -0
  69. reader_workbench/domains/plate_reader/analysis/four_state_event_window/reduction.py +62 -0
  70. reader_workbench/domains/plate_reader/analysis/four_state_event_window/seeds.py +15 -0
  71. reader_workbench/domains/plate_reader/analysis/four_state_event_window/sources.py +308 -0
  72. reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusion_validation.py +94 -0
  73. reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusions.py +54 -0
  74. reader_workbench/domains/plate_reader/analysis/timepoints.py +76 -0
  75. reader_workbench/domains/plate_reader/io/__init__.py +6 -0
  76. reader_workbench/domains/plate_reader/io/sample_map.py +65 -0
  77. reader_workbench/domains/plate_reader/io/synergy_h1/__init__.py +6 -0
  78. reader_workbench/domains/plate_reader/io/synergy_h1/_kinetic.py +141 -0
  79. reader_workbench/domains/plate_reader/io/synergy_h1/_parser.py +295 -0
  80. reader_workbench/domains/plate_reader/io/synergy_h1/_shared.py +195 -0
  81. reader_workbench/domains/plate_reader/io/synergy_h1/_snapshot.py +130 -0
  82. reader_workbench/domains/plate_reader/ordering.py +59 -0
  83. reader_workbench/domains/plate_reader/plots/__init__.py +15 -0
  84. reader_workbench/domains/plate_reader/plots/_data.py +29 -0
  85. reader_workbench/domains/plate_reader/plots/common.py +346 -0
  86. reader_workbench/domains/plate_reader/plots/distributions.py +324 -0
  87. reader_workbench/domains/plate_reader/plots/dual_reporter_triptych.py +525 -0
  88. reader_workbench/domains/plate_reader/plots/dual_reporter_triptych_render.py +195 -0
  89. reader_workbench/domains/plate_reader/plots/four_state_event_window/__init__.py +26 -0
  90. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic.py +295 -0
  91. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_components.py +149 -0
  92. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_render.py +308 -0
  93. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_style.py +51 -0
  94. reader_workbench/domains/plate_reader/plots/four_state_event_window/schema.py +8 -0
  95. reader_workbench/domains/plate_reader/plots/four_state_event_window/summary.py +140 -0
  96. reader_workbench/domains/plate_reader/plots/grouping.py +53 -0
  97. reader_workbench/domains/plate_reader/plots/panels/__init__.py +12 -0
  98. reader_workbench/domains/plate_reader/plots/panels/snapshot.py +161 -0
  99. reader_workbench/domains/plate_reader/plots/panels/snapshot_data.py +91 -0
  100. reader_workbench/domains/plate_reader/plots/panels/time_series.py +296 -0
  101. reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic.py +435 -0
  102. reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic_render.py +300 -0
  103. reader_workbench/domains/plate_reader/plots/snapshot_barplot/__init__.py +315 -0
  104. reader_workbench/domains/plate_reader/plots/snapshot_barplot/planning.py +168 -0
  105. reader_workbench/domains/plate_reader/plots/snapshot_heatmap/__init__.py +205 -0
  106. reader_workbench/domains/plate_reader/plots/snapshot_heatmap/inputs.py +103 -0
  107. reader_workbench/domains/plate_reader/plots/time_series.py +317 -0
  108. reader_workbench/domains/plate_reader/plots/ts_and_snap/__init__.py +479 -0
  109. reader_workbench/domains/plate_reader/plots/ts_and_snap/planning.py +283 -0
  110. reader_workbench/domains/time_series/__init__.py +29 -0
  111. reader_workbench/domains/time_series/aggregation.py +60 -0
  112. reader_workbench/domains/time_series/contracts.py +368 -0
  113. reader_workbench/domains/time_series/reduction.py +395 -0
  114. reader_workbench/errors.py +55 -0
  115. reader_workbench/maintenance/__init__.py +6 -0
  116. reader_workbench/maintenance/docs.py +335 -0
  117. reader_workbench/maintenance/model.py +28 -0
  118. reader_workbench/maintenance/release.py +39 -0
  119. reader_workbench/maintenance/skills.py +124 -0
  120. reader_workbench/plotting/__init__.py +20 -0
  121. reader_workbench/plotting/mpl.py +56 -0
  122. reader_workbench/plotting/sinks.py +69 -0
  123. reader_workbench/plotting/style.py +175 -0
  124. reader_workbench/plotting/utils.py +27 -0
  125. reader_workbench/plugins/__init__.py +1 -0
  126. reader_workbench/plugins/catalog.py +33 -0
  127. reader_workbench/plugins/export/__init__.py +0 -0
  128. reader_workbench/plugins/export/_paths.py +21 -0
  129. reader_workbench/plugins/export/csv.py +41 -0
  130. reader_workbench/plugins/export/xlsx.py +44 -0
  131. reader_workbench/plugins/ingest/__init__.py +0 -0
  132. reader_workbench/plugins/ingest/_discovery.py +58 -0
  133. reader_workbench/plugins/ingest/discovery_policy.py +66 -0
  134. reader_workbench/plugins/ingest/flow_cytometer.py +139 -0
  135. reader_workbench/plugins/ingest/synergy_h1.py +234 -0
  136. reader_workbench/plugins/manifests/__init__.py +1 -0
  137. reader_workbench/plugins/manifests/export.py +29 -0
  138. reader_workbench/plugins/manifests/ingest.py +29 -0
  139. reader_workbench/plugins/manifests/plot.py +161 -0
  140. reader_workbench/plugins/manifests/transform.py +172 -0
  141. reader_workbench/plugins/manifests/validator.py +18 -0
  142. reader_workbench/plugins/plot/__init__.py +0 -0
  143. reader_workbench/plugins/plot/_shared.py +55 -0
  144. reader_workbench/plugins/plot/cytometry_diagnostic.py +54 -0
  145. reader_workbench/plugins/plot/distributions.py +58 -0
  146. reader_workbench/plugins/plot/dual_reporter_triptych.py +224 -0
  147. reader_workbench/plugins/plot/four_state_event_window_diagnostic.py +133 -0
  148. reader_workbench/plugins/plot/four_state_event_window_summary.py +53 -0
  149. reader_workbench/plugins/plot/four_state_vector_collection.py +45 -0
  150. reader_workbench/plugins/plot/four_state_vector_diagnostic.py +105 -0
  151. reader_workbench/plugins/plot/four_state_vector_heatmap.py +63 -0
  152. reader_workbench/plugins/plot/logic_symmetry.py +56 -0
  153. reader_workbench/plugins/plot/single_reporter_diagnostic.py +279 -0
  154. reader_workbench/plugins/plot/snapshot_barplot.py +65 -0
  155. reader_workbench/plugins/plot/snapshot_heatmap.py +104 -0
  156. reader_workbench/plugins/plot/time_series.py +114 -0
  157. reader_workbench/plugins/plot/ts_and_snap.py +210 -0
  158. reader_workbench/plugins/transform/__init__.py +0 -0
  159. reader_workbench/plugins/transform/_four_state_vector.py +204 -0
  160. reader_workbench/plugins/transform/_labeling.py +109 -0
  161. reader_workbench/plugins/transform/alias.py +70 -0
  162. reader_workbench/plugins/transform/assay_labels.py +62 -0
  163. reader_workbench/plugins/transform/blank.py +79 -0
  164. reader_workbench/plugins/transform/crosstalk_pairs.py +180 -0
  165. reader_workbench/plugins/transform/cytometry_gating.py +120 -0
  166. reader_workbench/plugins/transform/fold_change.py +79 -0
  167. reader_workbench/plugins/transform/four_state_event_window.py +93 -0
  168. reader_workbench/plugins/transform/four_state_vector.py +62 -0
  169. reader_workbench/plugins/transform/four_state_vector_collection.py +41 -0
  170. reader_workbench/plugins/transform/logic_symmetry.py +67 -0
  171. reader_workbench/plugins/transform/outlier_filter.py +60 -0
  172. reader_workbench/plugins/transform/overflow.py +197 -0
  173. reader_workbench/plugins/transform/ratio.py +237 -0
  174. reader_workbench/plugins/transform/sample_map.py +170 -0
  175. reader_workbench/plugins/transform/sample_metadata.py +94 -0
  176. reader_workbench/plugins/validator/__init__.py +1 -0
  177. reader_workbench/plugins/validator/to_tidy_plus_map.py +155 -0
  178. reader_workbench/protocols/__init__.py +80 -0
  179. reader_workbench/protocols/_builtins_plate_reader_growth.py +179 -0
  180. reader_workbench/protocols/_builtins_plate_reader_variants.py +274 -0
  181. reader_workbench/protocols/builtins.py +1656 -0
  182. reader_workbench/protocols/compiler.py +22 -0
  183. reader_workbench/protocols/compilers/__init__.py +1 -0
  184. reader_workbench/protocols/compilers/common.py +100 -0
  185. reader_workbench/protocols/compilers/cytometry.py +87 -0
  186. reader_workbench/protocols/compilers/generic.py +14 -0
  187. reader_workbench/protocols/compilers/logic.py +245 -0
  188. reader_workbench/protocols/compilers/plate_reader.py +937 -0
  189. reader_workbench/protocols/compilers/plate_reader_pipeline.py +197 -0
  190. reader_workbench/protocols/model.py +1486 -0
  191. reader_workbench/protocols/semantic_coverage.py +234 -0
  192. reader_workbench/runtime/__init__.py +12 -0
  193. reader_workbench/runtime/builtin.py +23 -0
  194. reader_workbench/runtime/model.py +42 -0
  195. reader_workbench/workbench/__init__.py +60 -0
  196. reader_workbench/workbench/assets/__init__.py +22 -0
  197. reader_workbench/workbench/assets/types.py +118 -0
  198. reader_workbench/workbench/audit/__init__.py +5 -0
  199. reader_workbench/workbench/audit/experiments.py +307 -0
  200. reader_workbench/workbench/audit/staging.py +187 -0
  201. reader_workbench/workbench/cli/__init__.py +51 -0
  202. reader_workbench/workbench/cli/_lazy.py +9 -0
  203. reader_workbench/workbench/cli/_records_view.py +150 -0
  204. reader_workbench/workbench/cli/_surface_execution.py +443 -0
  205. reader_workbench/workbench/cli/audit.py +95 -0
  206. reader_workbench/workbench/cli/automation.py +229 -0
  207. reader_workbench/workbench/cli/demo.py +46 -0
  208. reader_workbench/workbench/cli/dop.py +91 -0
  209. reader_workbench/workbench/cli/experiments.py +635 -0
  210. reader_workbench/workbench/cli/helpers.py +232 -0
  211. reader_workbench/workbench/cli/main.py +59 -0
  212. reader_workbench/workbench/cli/maintenance.py +82 -0
  213. reader_workbench/workbench/cli/notebooks.py +260 -0
  214. reader_workbench/workbench/cli/pagination.py +117 -0
  215. reader_workbench/workbench/cli/protocols.py +336 -0
  216. reader_workbench/workbench/cli/shared.py +309 -0
  217. reader_workbench/workbench/cli/surfaces.py +534 -0
  218. reader_workbench/workbench/cli/verification.py +128 -0
  219. reader_workbench/workbench/commands.py +10 -0
  220. reader_workbench/workbench/config/__init__.py +47 -0
  221. reader_workbench/workbench/config/identity.py +13 -0
  222. reader_workbench/workbench/config/load.py +405 -0
  223. reader_workbench/workbench/config/model.py +274 -0
  224. reader_workbench/workbench/context.py +26 -0
  225. reader_workbench/workbench/decl/__init__.py +31 -0
  226. reader_workbench/workbench/decl/build.py +190 -0
  227. reader_workbench/workbench/decl/model.py +81 -0
  228. reader_workbench/workbench/dop/__init__.py +12 -0
  229. reader_workbench/workbench/dop/builtins.py +261 -0
  230. reader_workbench/workbench/dop/model.py +209 -0
  231. reader_workbench/workbench/engine/__init__.py +42 -0
  232. reader_workbench/workbench/engine/_shared.py +76 -0
  233. reader_workbench/workbench/engine/contracts.py +283 -0
  234. reader_workbench/workbench/engine/execution.py +326 -0
  235. reader_workbench/workbench/engine/file_outputs.py +260 -0
  236. reader_workbench/workbench/engine/inputs.py +161 -0
  237. reader_workbench/workbench/engine/invocations.py +507 -0
  238. reader_workbench/workbench/engine/planning.py +72 -0
  239. reader_workbench/workbench/engine/runtime.py +464 -0
  240. reader_workbench/workbench/engine/setup.py +149 -0
  241. reader_workbench/workbench/engine/validation.py +684 -0
  242. reader_workbench/workbench/experiment/__init__.py +47 -0
  243. reader_workbench/workbench/experiment/model.py +381 -0
  244. reader_workbench/workbench/experiments.py +133 -0
  245. reader_workbench/workbench/graph/__init__.py +47 -0
  246. reader_workbench/workbench/graph/nodes.py +102 -0
  247. reader_workbench/workbench/graph/normalize.py +177 -0
  248. reader_workbench/workbench/graph/refs.py +148 -0
  249. reader_workbench/workbench/input_discovery.py +19 -0
  250. reader_workbench/workbench/inspection/__init__.py +3 -0
  251. reader_workbench/workbench/inspection/catalogs.py +128 -0
  252. reader_workbench/workbench/inspection/common.py +92 -0
  253. reader_workbench/workbench/inspection/dop.py +64 -0
  254. reader_workbench/workbench/inspection/experiments.py +449 -0
  255. reader_workbench/workbench/inspection/inventory.py +68 -0
  256. reader_workbench/workbench/inspection/protocols.py +368 -0
  257. reader_workbench/workbench/inspection/readiness.py +333 -0
  258. reader_workbench/workbench/inspection/reports.py +367 -0
  259. reader_workbench/workbench/inspection/results.py +166 -0
  260. reader_workbench/workbench/inspection/runtime.py +287 -0
  261. reader_workbench/workbench/inspection/semantics.py +192 -0
  262. reader_workbench/workbench/inspection/validation.py +30 -0
  263. reader_workbench/workbench/notebooks/__init__.py +17 -0
  264. reader_workbench/workbench/notebooks/_launch_registry.py +112 -0
  265. reader_workbench/workbench/notebooks/_launch_runtime.py +104 -0
  266. reader_workbench/workbench/notebooks/components/__init__.py +21 -0
  267. reader_workbench/workbench/notebooks/components/deliverables.py +403 -0
  268. reader_workbench/workbench/notebooks/components/overview.py +119 -0
  269. reader_workbench/workbench/notebooks/eda.marimo.py.txt +153 -0
  270. reader_workbench/workbench/notebooks/launch.py +274 -0
  271. reader_workbench/workbench/notebooks/presentation.py +136 -0
  272. reader_workbench/workbench/notebooks/scaffold.py +60 -0
  273. reader_workbench/workbench/ontology.py +78 -0
  274. reader_workbench/workbench/paths.py +44 -0
  275. reader_workbench/workbench/ports/__init__.py +31 -0
  276. reader_workbench/workbench/ports/model.py +168 -0
  277. reader_workbench/workbench/records/__init__.py +44 -0
  278. reader_workbench/workbench/records/epoch.py +329 -0
  279. reader_workbench/workbench/records/evidence.py +247 -0
  280. reader_workbench/workbench/records/identity.py +87 -0
  281. reader_workbench/workbench/records/locking.py +185 -0
  282. reader_workbench/workbench/records/model.py +711 -0
  283. reader_workbench/workbench/records/sources.py +73 -0
  284. reader_workbench/workbench/records/store.py +1022 -0
  285. reader_workbench/workbench/records/verification.py +998 -0
  286. reader_workbench/workbench/registry.py +333 -0
  287. reader_workbench/workbench/spec_overrides.py +215 -0
  288. reader_workbench-1.0.0.dist-info/METADATA +91 -0
  289. reader_workbench-1.0.0.dist-info/RECORD +293 -0
  290. reader_workbench-1.0.0.dist-info/WHEEL +5 -0
  291. reader_workbench-1.0.0.dist-info/entry_points.txt +2 -0
  292. reader_workbench-1.0.0.dist-info/licenses/LICENSE +21 -0
  293. reader_workbench-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,47 @@
1
+ from .identity import reader_spec_digest
2
+ from .load import load_reader_config_document
3
+ from .model import (
4
+ AnnotationCollectionSpec,
5
+ AnnotationLabelSpec,
6
+ AnnotationOrderedStateSpaceSpec,
7
+ AnnotationOrderSpec,
8
+ AnnotationSpec,
9
+ EvidenceSpec,
10
+ ExperimentSpec,
11
+ ExportOutputsSpec,
12
+ FileResourceSpec,
13
+ OutputsSpec,
14
+ PathsSpec,
15
+ PlotOutputsSpec,
16
+ PlottingSpec,
17
+ ProtocolBindingSpec,
18
+ ReaderSpec,
19
+ RecordResourceSpec,
20
+ ReplicateKind,
21
+ ResourceSpec,
22
+ ResourcesSpec,
23
+ )
24
+
25
+ __all__ = [
26
+ "AnnotationCollectionSpec",
27
+ "AnnotationLabelSpec",
28
+ "AnnotationOrderedStateSpaceSpec",
29
+ "AnnotationOrderSpec",
30
+ "AnnotationSpec",
31
+ "EvidenceSpec",
32
+ "ExportOutputsSpec",
33
+ "ExperimentSpec",
34
+ "FileResourceSpec",
35
+ "OutputsSpec",
36
+ "PathsSpec",
37
+ "PlotOutputsSpec",
38
+ "PlottingSpec",
39
+ "ProtocolBindingSpec",
40
+ "ReplicateKind",
41
+ "ReaderSpec",
42
+ "RecordResourceSpec",
43
+ "ResourceSpec",
44
+ "ResourcesSpec",
45
+ "load_reader_config_document",
46
+ "reader_spec_digest",
47
+ ]
@@ -0,0 +1,13 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+
6
+ from .model import ReaderSpec
7
+
8
+
9
+ def reader_spec_digest(spec: ReaderSpec) -> str:
10
+ """Digest the normalized, complete reader/v8 experiment configuration."""
11
+ payload = spec.model_dump(mode="json", by_alias=True, exclude_none=False)
12
+ raw = json.dumps(payload, ensure_ascii=True, sort_keys=True, separators=(",", ":")).encode("utf-8")
13
+ return "sha256:" + hashlib.sha256(raw).hexdigest()
@@ -0,0 +1,405 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+ from typing import Any
5
+
6
+ import yaml
7
+ from pydantic import ValidationError
8
+ from yaml.constructor import ConstructorError
9
+ from yaml.resolver import BaseResolver
10
+
11
+ from reader_workbench.errors import ConfigError
12
+
13
+ from .model import ReaderSpec
14
+
15
+
16
+ class _UniqueKeyLoader(yaml.SafeLoader):
17
+ """Safe YAML loader that rejects duplicate mapping keys."""
18
+
19
+
20
+ def _construct_unique_mapping(loader: _UniqueKeyLoader, node: yaml.MappingNode, deep: bool = False):
21
+ mapping: dict[Any, Any] = {}
22
+ for key_node, value_node in node.value:
23
+ key = loader.construct_object(key_node, deep=deep)
24
+ if key in mapping:
25
+ raise ConstructorError(
26
+ "while constructing a mapping",
27
+ node.start_mark,
28
+ f"found duplicate key {key!r}",
29
+ key_node.start_mark,
30
+ )
31
+ mapping[key] = loader.construct_object(value_node, deep=deep)
32
+ return mapping
33
+
34
+
35
+ _UniqueKeyLoader.add_constructor(BaseResolver.DEFAULT_MAPPING_TAG, _construct_unique_mapping)
36
+
37
+
38
+ def load_reader_config_document(path: Path) -> dict[str, Any]:
39
+ """Load a Reader config document with strict YAML and schema identity checks."""
40
+
41
+ try:
42
+ text = path.read_text(encoding="utf-8")
43
+ except UnicodeError as exc:
44
+ raise ConfigError(f"Could not read UTF-8 config {path}: {exc}") from exc
45
+ except OSError as exc:
46
+ raise ConfigError(f"Could not read config {path}: {exc}") from exc
47
+ try:
48
+ data = yaml.load(text, Loader=_UniqueKeyLoader)
49
+ except yaml.YAMLError as exc:
50
+ raise ConfigError(f"Invalid YAML in {path}: {exc}") from exc
51
+ if not isinstance(data, dict):
52
+ raise ConfigError(
53
+ f"Config must be a mapping (YAML object) in {path}. Check for empty files or top-level lists."
54
+ )
55
+
56
+ schema = data.get("schema")
57
+ if schema != "reader/v8":
58
+ raise ConfigError(f"Config schema must be 'reader/v8'. This repo only supports reader/v8 (found {schema!r}).")
59
+ return data
60
+
61
+
62
+ def load_reader_spec(path: Path, *, cls: type[ReaderSpec]) -> ReaderSpec:
63
+ data = load_reader_config_document(path)
64
+
65
+ removed_top_level_keys = {
66
+ "steps",
67
+ "overrides",
68
+ "collections",
69
+ "graph_patch",
70
+ "deliverable_presets",
71
+ "deliverable_overrides",
72
+ "notebook",
73
+ "data",
74
+ "semantics",
75
+ "assay",
76
+ "pipeline",
77
+ "plots",
78
+ "exports",
79
+ "notebooks",
80
+ }
81
+ removed_protocol_keys = {"with", "plugins", "parameters", "deliverables"}
82
+ removed_experiment_keys = {"name", "outputs", "plots_dir", "palette"}
83
+ illegal = sorted(key for key in removed_top_level_keys if key in data)
84
+ illegal_protocol = []
85
+ if "protocol" in data and isinstance(data["protocol"], dict):
86
+ illegal_protocol = sorted(key for key in removed_protocol_keys if key in data["protocol"])
87
+ illegal_exp = []
88
+ if "experiment" in data and isinstance(data["experiment"], dict):
89
+ illegal_exp = sorted(key for key in removed_experiment_keys if key in data["experiment"])
90
+ if illegal or illegal_protocol or illegal_exp:
91
+ parts = []
92
+ if illegal:
93
+ parts.append(f"top-level keys: {illegal}")
94
+ if illegal_protocol:
95
+ parts.append(f"protocol keys: {illegal_protocol}")
96
+ if illegal_exp:
97
+ parts.append(f"experiment keys: {illegal_exp}")
98
+ raise ConfigError("reader/v8 rejects removed config keys. Remove or replace: " + "; ".join(parts))
99
+
100
+ data.setdefault("experiment", {})
101
+ if not isinstance(data["experiment"], dict):
102
+ raise ConfigError("experiment must be a mapping when provided")
103
+ experiment_id = data["experiment"].get("id")
104
+ if not isinstance(experiment_id, str) or not experiment_id.strip():
105
+ raise ConfigError("experiment.id is required and must be a non-empty string")
106
+ data["experiment"].setdefault("title", experiment_id)
107
+
108
+ protocol = data.get("protocol")
109
+ if not isinstance(protocol, dict):
110
+ raise ConfigError("protocol is required and must be a mapping with id/inputs/analysis/outputs")
111
+ _ensure_only_keys(protocol, {"id", "inputs", "analysis", "outputs"}, where="protocol")
112
+ protocol_id = protocol.get("id")
113
+ if not isinstance(protocol_id, str) or not protocol_id.strip():
114
+ raise ConfigError("protocol.id must be a non-empty string")
115
+ inputs = protocol.get("inputs", {}) or {}
116
+ analysis = protocol.get("analysis", {}) or {}
117
+ outputs = protocol.get("outputs", {}) or {}
118
+ if not isinstance(inputs, dict):
119
+ raise ConfigError("protocol.inputs must be a mapping")
120
+ if not isinstance(analysis, dict):
121
+ raise ConfigError("protocol.analysis must be a mapping")
122
+ if not isinstance(outputs, dict):
123
+ raise ConfigError("protocol.outputs must be a mapping")
124
+ data["protocol"] = {
125
+ "id": protocol_id.strip(),
126
+ "inputs": dict(inputs),
127
+ "analysis": dict(analysis),
128
+ "outputs": _normalize_outputs(outputs),
129
+ }
130
+
131
+ data.setdefault("paths", {})
132
+ if not isinstance(data["paths"], dict):
133
+ raise ConfigError("paths must be a mapping")
134
+ outputs_raw = data["paths"].get("outputs", "./outputs")
135
+ if not isinstance(outputs_raw, str) or not outputs_raw.strip():
136
+ raise ConfigError("paths.outputs must be a non-empty string path")
137
+ data["paths"]["outputs"] = outputs_raw
138
+ for key, value in (
139
+ ("plots", data["paths"].get("plots", "plots")),
140
+ ("exports", data["paths"].get("exports", "exports")),
141
+ ("notebooks", data["paths"].get("notebooks", "notebooks")),
142
+ ):
143
+ if value is None:
144
+ raise ConfigError(f"paths.{key} must be a string subdirectory (use '.' to flatten).")
145
+ if not isinstance(value, str):
146
+ raise ConfigError(f"paths.{key} must be a string subdirectory")
147
+ subdir = Path(value)
148
+ if subdir.is_absolute():
149
+ raise ConfigError(f"paths.{key} must be relative to paths.outputs, not absolute.")
150
+ normalized_subdir = (Path(".") / subdir).parts
151
+ if ".." in normalized_subdir:
152
+ raise ConfigError(f"paths.{key} must stay under paths.outputs and may not escape via '..'.")
153
+ data["paths"][key] = value
154
+
155
+ data.setdefault("plotting", {})
156
+ if not isinstance(data["plotting"], dict):
157
+ raise ConfigError("plotting must be a mapping")
158
+ palette_raw = data["plotting"].get("palette", None)
159
+ if palette_raw is not None and (not isinstance(palette_raw, str) or not palette_raw.strip()):
160
+ raise ConfigError("plotting.palette must be a non-empty string or null")
161
+
162
+ data.setdefault("resources", {})
163
+ if not isinstance(data["resources"], dict):
164
+ raise ConfigError("resources must be a mapping of resource_id -> file or record declarations")
165
+ normalized_resources: dict[str, dict[str, str]] = {}
166
+ for resource_id, resource in (data["resources"] or {}).items():
167
+ if not isinstance(resource, dict):
168
+ raise ConfigError(f"resources.{resource_id} must be a mapping")
169
+ kind = resource.get("kind")
170
+ if kind == "file":
171
+ _ensure_only_keys(resource, {"kind", "path"}, where=f"resources.{resource_id}")
172
+ path_raw = resource.get("path")
173
+ if not isinstance(path_raw, str) or not path_raw.strip():
174
+ raise ConfigError(f"resources.{resource_id}.path must be a non-empty string")
175
+ normalized_resources[str(resource_id)] = {"kind": "file", "path": path_raw}
176
+ continue
177
+ if kind == "record":
178
+ _ensure_only_keys(resource, {"kind", "experiment", "record"}, where=f"resources.{resource_id}")
179
+ experiment = resource.get("experiment")
180
+ record = resource.get("record")
181
+ if not isinstance(experiment, str) or not experiment.strip():
182
+ raise ConfigError(f"resources.{resource_id}.experiment must be a non-empty string")
183
+ if not isinstance(record, str) or not record.strip():
184
+ raise ConfigError(f"resources.{resource_id}.record must be a non-empty string")
185
+ normalized_resources[str(resource_id)] = {
186
+ "kind": "record",
187
+ "experiment": experiment.strip(),
188
+ "record": record.strip(),
189
+ }
190
+ continue
191
+ raise ConfigError(f"resources.{resource_id}.kind must be 'file' or 'record'")
192
+ data["resources"] = {"by_id": normalized_resources}
193
+
194
+ data.setdefault("annotations", {})
195
+ if not isinstance(data["annotations"], dict):
196
+ raise ConfigError("annotations must be a mapping")
197
+ data["annotations"] = _normalize_annotations(data["annotations"])
198
+
199
+ try:
200
+ return cls.model_validate(data)
201
+ except ValidationError as exc:
202
+ raise ConfigError(str(exc)) from exc
203
+
204
+
205
+ def _normalize_annotations(annotations_raw: dict[str, Any]) -> dict[str, Any]:
206
+ _ensure_only_keys(
207
+ annotations_raw,
208
+ {"labels", "orders", "collections", "ordered_state_spaces"},
209
+ where="annotations",
210
+ )
211
+ labels_raw = annotations_raw.get("labels", {}) or {}
212
+ if not isinstance(labels_raw, dict):
213
+ raise ConfigError("annotations.labels must be a mapping")
214
+ normalized_labels: dict[str, dict[str, Any]] = {}
215
+ for label_id, label_spec in labels_raw.items():
216
+ if not isinstance(label_spec, dict):
217
+ raise ConfigError(f"annotations.labels.{label_id} must be a mapping")
218
+ _ensure_only_keys(label_spec, {"source", "values", "output"}, where=f"annotations.labels.{label_id}")
219
+ source = label_spec.get("source")
220
+ if not isinstance(source, str) or not source.strip():
221
+ raise ConfigError(f"annotations.labels.{label_id}.source must be a non-empty string")
222
+ values = label_spec.get("values", {}) or {}
223
+ if not isinstance(values, dict):
224
+ raise ConfigError(f"annotations.labels.{label_id}.values must be a mapping")
225
+ output = label_spec.get("output")
226
+ if output is not None and (not isinstance(output, str) or not output.strip()):
227
+ raise ConfigError(f"annotations.labels.{label_id}.output must be a non-empty string when provided")
228
+ normalized_labels[str(label_id)] = {
229
+ "source": source,
230
+ "values": {str(k): str(v) for k, v in values.items()},
231
+ "output": (str(output) if isinstance(output, str) else None),
232
+ }
233
+
234
+ orders_raw = annotations_raw.get("orders", {}) or {}
235
+ if not isinstance(orders_raw, dict):
236
+ raise ConfigError("annotations.orders must be a mapping")
237
+ normalized_orders: dict[str, dict[str, Any]] = {}
238
+ for order_id, order_spec in orders_raw.items():
239
+ if not isinstance(order_spec, dict):
240
+ raise ConfigError(f"annotations.orders.{order_id} must be a mapping")
241
+ _ensure_only_keys(order_spec, {"column", "values"}, where=f"annotations.orders.{order_id}")
242
+ column = order_spec.get("column")
243
+ if not isinstance(column, str) or not column.strip():
244
+ raise ConfigError(f"annotations.orders.{order_id}.column must be a non-empty string")
245
+ values = order_spec.get("values", []) or []
246
+ if not isinstance(values, list) or any(isinstance(item, (dict, list)) for item in values):
247
+ raise ConfigError(f"annotations.orders.{order_id}.values must be a flat list of scalar labels")
248
+ if not values:
249
+ raise ConfigError(f"annotations.orders.{order_id}.values must not be empty")
250
+ normalized_orders[str(order_id)] = {"column": column, "values": [str(item) for item in values]}
251
+
252
+ collections_raw = annotations_raw.get("collections", {}) or {}
253
+ if not isinstance(collections_raw, dict):
254
+ raise ConfigError("annotations.collections must be a mapping")
255
+ normalized_collections: dict[str, dict[str, Any]] = {}
256
+ for collection_id, collection_spec in collections_raw.items():
257
+ if not isinstance(collection_spec, dict):
258
+ raise ConfigError(f"annotations.collections.{collection_id} must be a mapping")
259
+ _ensure_only_keys(collection_spec, {"column", "items"}, where=f"annotations.collections.{collection_id}")
260
+ column = collection_spec.get("column")
261
+ if not isinstance(column, str) or not column.strip():
262
+ raise ConfigError(f"annotations.collections.{collection_id}.column must be a non-empty string")
263
+ items = collection_spec.get("items", {}) or {}
264
+ if not isinstance(items, dict):
265
+ raise ConfigError(f"annotations.collections.{collection_id}.items must be a mapping")
266
+ invalid_items = sorted(item_key for item_key, item_values in items.items() if not isinstance(item_values, list))
267
+ if invalid_items:
268
+ raise ConfigError(
269
+ f"annotations.collections.{collection_id}.items entries must be lists for keys: {invalid_items}"
270
+ )
271
+ normalized_collections[str(collection_id)] = {
272
+ "column": column,
273
+ "items": {str(item_key): [str(item) for item in item_values] for item_key, item_values in items.items()},
274
+ }
275
+
276
+ state_spaces_raw = annotations_raw.get("ordered_state_spaces", {}) or {}
277
+ if not isinstance(state_spaces_raw, dict):
278
+ raise ConfigError("annotations.ordered_state_spaces must be a mapping")
279
+ normalized_state_spaces: dict[str, dict[str, Any]] = {}
280
+ for space_id, space_spec in state_spaces_raw.items():
281
+ if not isinstance(space_id, str) or not space_id or space_id != space_id.strip():
282
+ raise ConfigError("annotations.ordered_state_spaces keys must be non-empty, already-trimmed strings")
283
+ context = f"annotations.ordered_state_spaces.{space_id}"
284
+ if not isinstance(space_spec, dict):
285
+ raise ConfigError(f"{context} must be a mapping")
286
+ _ensure_only_keys(space_spec, {"column", "state_order", "values", "case_sensitive"}, where=context)
287
+ column = space_spec.get("column")
288
+ if not isinstance(column, str) or not column.strip():
289
+ raise ConfigError(f"{context}.column must be a non-empty string")
290
+ state_order = space_spec.get("state_order")
291
+ if not isinstance(state_order, list) or not state_order:
292
+ raise ConfigError(f"{context}.state_order must be a non-empty list")
293
+ if any(
294
+ not isinstance(state_id, str) or not state_id or state_id != state_id.strip() for state_id in state_order
295
+ ):
296
+ raise ConfigError(f"{context}.state_order must contain non-empty, already-trimmed strings")
297
+ normalized_state_order = list(state_order)
298
+ if len(set(normalized_state_order)) != len(normalized_state_order):
299
+ raise ConfigError(f"{context} state ids must be unique")
300
+ values = space_spec.get("values")
301
+ if not isinstance(values, dict) or not values:
302
+ raise ConfigError(f"{context}.values must be a non-empty mapping")
303
+ if any(not isinstance(state_id, str) for state_id in values):
304
+ raise ConfigError(f"{context}.values keys must be strings")
305
+ if any(not isinstance(value, str) or not value.strip() for value in values.values()):
306
+ raise ConfigError(f"{context}.values must map state ids to non-empty strings")
307
+ normalized_values = dict(values)
308
+ if set(normalized_values) != set(normalized_state_order):
309
+ raise ConfigError(f"{context}.values must have exactly the ids declared by state_order")
310
+ case_sensitive = space_spec.get("case_sensitive", True)
311
+ if not isinstance(case_sensitive, bool):
312
+ raise ConfigError(f"{context}.case_sensitive must be a boolean")
313
+ comparison_values = [
314
+ normalized_values[state_id] if case_sensitive else normalized_values[state_id].strip().casefold()
315
+ for state_id in normalized_state_order
316
+ ]
317
+ if len(set(comparison_values)) != len(comparison_values):
318
+ sensitivity = "true" if case_sensitive else "false"
319
+ raise ConfigError(f"{context} source values must be unique under case_sensitive={sensitivity}")
320
+ normalized_state_spaces[space_id] = {
321
+ "column": column.strip(),
322
+ "state_order": normalized_state_order,
323
+ "values": normalized_values,
324
+ "case_sensitive": case_sensitive,
325
+ }
326
+
327
+ return {
328
+ "labels": normalized_labels,
329
+ "orders": normalized_orders,
330
+ "collections": normalized_collections,
331
+ "ordered_state_spaces": normalized_state_spaces,
332
+ }
333
+
334
+
335
+ def _normalize_outputs(raw: dict[str, Any]) -> dict[str, Any]:
336
+ normalized: dict[str, Any] = {}
337
+ _ensure_only_keys(raw, {"plots", "exports"}, where="protocol.outputs")
338
+
339
+ plots = raw.get("plots", {}) or {}
340
+ if not isinstance(plots, dict):
341
+ raise ConfigError("protocol.outputs.plots must be a mapping")
342
+ _ensure_only_keys(plots, {"profile", "include", "exclude", "views"}, where="protocol.outputs.plots")
343
+ plot_profile = plots.get("profile")
344
+ if plot_profile is not None and (not isinstance(plot_profile, str) or not plot_profile.strip()):
345
+ raise ConfigError("protocol.outputs.plots.profile must be a non-empty string when provided")
346
+ plot_include = plots.get("include", []) or []
347
+ plot_exclude = plots.get("exclude", []) or []
348
+ plot_views = plots.get("views", {}) or {}
349
+ if not isinstance(plot_include, list) or not all(isinstance(item, str) and item.strip() for item in plot_include):
350
+ raise ConfigError("protocol.outputs.plots.include must be a list of non-empty plot ids")
351
+ if not isinstance(plot_exclude, list) or not all(isinstance(item, str) and item.strip() for item in plot_exclude):
352
+ raise ConfigError("protocol.outputs.plots.exclude must be a list of non-empty plot ids")
353
+ if not isinstance(plot_views, dict):
354
+ raise ConfigError("protocol.outputs.plots.views must be a mapping of plot id -> view config")
355
+ normalized["plots"] = {
356
+ "profile": plot_profile,
357
+ "include": [str(item) for item in plot_include],
358
+ "exclude": [str(item) for item in plot_exclude],
359
+ "views": {
360
+ str(plot_id): _normalize_mapping(settings_block, where="protocol.outputs.plots.views")
361
+ for plot_id, settings_block in plot_views.items()
362
+ },
363
+ }
364
+
365
+ exports = raw.get("exports", {}) or {}
366
+ if not isinstance(exports, dict):
367
+ raise ConfigError("protocol.outputs.exports must be a mapping")
368
+ _ensure_only_keys(exports, {"include", "exclude", "artifacts"}, where="protocol.outputs.exports")
369
+ export_include = exports.get("include", []) or []
370
+ export_exclude = exports.get("exclude", []) or []
371
+ export_artifacts = exports.get("artifacts", {}) or {}
372
+ if not isinstance(export_include, list) or not all(
373
+ isinstance(item, str) and item.strip() for item in export_include
374
+ ):
375
+ raise ConfigError("protocol.outputs.exports.include must be a list of non-empty artifact ids")
376
+ if not isinstance(export_exclude, list) or not all(
377
+ isinstance(item, str) and item.strip() for item in export_exclude
378
+ ):
379
+ raise ConfigError("protocol.outputs.exports.exclude must be a list of non-empty artifact ids")
380
+ if not isinstance(export_artifacts, dict):
381
+ raise ConfigError("protocol.outputs.exports.artifacts must be a mapping of artifact id -> config")
382
+ normalized["exports"] = {
383
+ "include": [str(item) for item in export_include],
384
+ "exclude": [str(item) for item in export_exclude],
385
+ "artifacts": {
386
+ str(artifact_id): _normalize_mapping(settings_block, where="protocol.outputs.exports.artifacts")
387
+ for artifact_id, settings_block in export_artifacts.items()
388
+ },
389
+ }
390
+ return normalized
391
+
392
+
393
+ def _normalize_mapping(raw: Any, *, where: str) -> dict[str, Any]:
394
+ if raw is None:
395
+ return {}
396
+ if not isinstance(raw, dict):
397
+ raise ConfigError(f"{where} entries must be mappings")
398
+ return dict(raw)
399
+
400
+
401
+ def _ensure_only_keys(raw: dict[str, Any], allowed: set[str], *, where: str) -> None:
402
+ unknown = sorted(key for key in raw if key not in allowed)
403
+ if unknown:
404
+ options = ", ".join(sorted(allowed)) or "—"
405
+ raise ConfigError(f"{where} has unknown keys {unknown}. Allowed keys: {options}")