reader-workbench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (293) hide show
  1. reader_workbench/__init__.py +22 -0
  2. reader_workbench/__main__.py +4 -0
  3. reader_workbench/_version.py +17 -0
  4. reader_workbench/api/__init__.py +74 -0
  5. reader_workbench/api/_record_reads.py +75 -0
  6. reader_workbench/api/artifacts.py +79 -0
  7. reader_workbench/api/facade.py +538 -0
  8. reader_workbench/api/models.py +285 -0
  9. reader_workbench/api/notebooks.py +63 -0
  10. reader_workbench/contracts/__init__.py +18 -0
  11. reader_workbench/contracts/builtins/__init__.py +36 -0
  12. reader_workbench/contracts/builtins/cytometry.py +140 -0
  13. reader_workbench/contracts/builtins/four_state_event_window.py +214 -0
  14. reader_workbench/contracts/builtins/generic.py +21 -0
  15. reader_workbench/contracts/builtins/logic.py +149 -0
  16. reader_workbench/contracts/builtins/plate_reader.py +47 -0
  17. reader_workbench/contracts/catalog.py +257 -0
  18. reader_workbench/contracts/model.py +109 -0
  19. reader_workbench/domains/__init__.py +1 -0
  20. reader_workbench/domains/cytometry/__init__.py +3 -0
  21. reader_workbench/domains/cytometry/analysis/__init__.py +29 -0
  22. reader_workbench/domains/cytometry/analysis/events.py +182 -0
  23. reader_workbench/domains/cytometry/analysis/gating.py +175 -0
  24. reader_workbench/domains/cytometry/analysis/workflow.py +248 -0
  25. reader_workbench/domains/cytometry/io/__init__.py +3 -0
  26. reader_workbench/domains/cytometry/io/fcs.py +135 -0
  27. reader_workbench/domains/cytometry/plots/__init__.py +5 -0
  28. reader_workbench/domains/cytometry/plots/diagnostic.py +155 -0
  29. reader_workbench/domains/logic/__init__.py +3 -0
  30. reader_workbench/domains/logic/crosstalk/__init__.py +3 -0
  31. reader_workbench/domains/logic/crosstalk/pairs.py +661 -0
  32. reader_workbench/domains/logic/four_state_vector/__init__.py +7 -0
  33. reader_workbench/domains/logic/four_state_vector/builder.py +321 -0
  34. reader_workbench/domains/logic/four_state_vector/collection/__init__.py +20 -0
  35. reader_workbench/domains/logic/four_state_vector/collection/checks.py +49 -0
  36. reader_workbench/domains/logic/four_state_vector/collection/constants.py +29 -0
  37. reader_workbench/domains/logic/four_state_vector/collection/model.py +19 -0
  38. reader_workbench/domains/logic/four_state_vector/collection/render.py +383 -0
  39. reader_workbench/domains/logic/four_state_vector/collection/sources.py +185 -0
  40. reader_workbench/domains/logic/four_state_vector/config.py +214 -0
  41. reader_workbench/domains/logic/four_state_vector/diagnostic.py +361 -0
  42. reader_workbench/domains/logic/four_state_vector/heatmap.py +86 -0
  43. reader_workbench/domains/logic/four_state_vector/math.py +191 -0
  44. reader_workbench/domains/logic/four_state_vector/reference.py +85 -0
  45. reader_workbench/domains/logic/four_state_vector/selection.py +228 -0
  46. reader_workbench/domains/logic/four_state_vector/treatment_semantics.py +51 -0
  47. reader_workbench/domains/logic/four_state_vector/validation.py +19 -0
  48. reader_workbench/domains/logic/logic_symmetry/__init__.py +3 -0
  49. reader_workbench/domains/logic/logic_symmetry/encodings.py +93 -0
  50. reader_workbench/domains/logic/logic_symmetry/extract_corners.py +192 -0
  51. reader_workbench/domains/logic/logic_symmetry/main.py +236 -0
  52. reader_workbench/domains/logic/logic_symmetry/metrics.py +96 -0
  53. reader_workbench/domains/logic/logic_symmetry/overlay.py +129 -0
  54. reader_workbench/domains/logic/logic_symmetry/prep.py +138 -0
  55. reader_workbench/domains/logic/logic_symmetry/render.py +356 -0
  56. reader_workbench/domains/logic/treatment_columns.py +42 -0
  57. reader_workbench/domains/plate_reader/__init__.py +1 -0
  58. reader_workbench/domains/plate_reader/analysis/__init__.py +14 -0
  59. reader_workbench/domains/plate_reader/analysis/fold_change.py +474 -0
  60. reader_workbench/domains/plate_reader/analysis/four_state_event_window/__init__.py +21 -0
  61. reader_workbench/domains/plate_reader/analysis/four_state_event_window/aggregation.py +191 -0
  62. reader_workbench/domains/plate_reader/analysis/four_state_event_window/contract_fields.py +50 -0
  63. reader_workbench/domains/plate_reader/analysis/four_state_event_window/contracts.py +320 -0
  64. reader_workbench/domains/plate_reader/analysis/four_state_event_window/design_dispositions.py +54 -0
  65. reader_workbench/domains/plate_reader/analysis/four_state_event_window/disposition_records.py +140 -0
  66. reader_workbench/domains/plate_reader/analysis/four_state_event_window/event_sensitivity.py +27 -0
  67. reader_workbench/domains/plate_reader/analysis/four_state_event_window/materialize.py +274 -0
  68. reader_workbench/domains/plate_reader/analysis/four_state_event_window/observation_resampling.py +97 -0
  69. reader_workbench/domains/plate_reader/analysis/four_state_event_window/reduction.py +62 -0
  70. reader_workbench/domains/plate_reader/analysis/four_state_event_window/seeds.py +15 -0
  71. reader_workbench/domains/plate_reader/analysis/four_state_event_window/sources.py +308 -0
  72. reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusion_validation.py +94 -0
  73. reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusions.py +54 -0
  74. reader_workbench/domains/plate_reader/analysis/timepoints.py +76 -0
  75. reader_workbench/domains/plate_reader/io/__init__.py +6 -0
  76. reader_workbench/domains/plate_reader/io/sample_map.py +65 -0
  77. reader_workbench/domains/plate_reader/io/synergy_h1/__init__.py +6 -0
  78. reader_workbench/domains/plate_reader/io/synergy_h1/_kinetic.py +141 -0
  79. reader_workbench/domains/plate_reader/io/synergy_h1/_parser.py +295 -0
  80. reader_workbench/domains/plate_reader/io/synergy_h1/_shared.py +195 -0
  81. reader_workbench/domains/plate_reader/io/synergy_h1/_snapshot.py +130 -0
  82. reader_workbench/domains/plate_reader/ordering.py +59 -0
  83. reader_workbench/domains/plate_reader/plots/__init__.py +15 -0
  84. reader_workbench/domains/plate_reader/plots/_data.py +29 -0
  85. reader_workbench/domains/plate_reader/plots/common.py +346 -0
  86. reader_workbench/domains/plate_reader/plots/distributions.py +324 -0
  87. reader_workbench/domains/plate_reader/plots/dual_reporter_triptych.py +525 -0
  88. reader_workbench/domains/plate_reader/plots/dual_reporter_triptych_render.py +195 -0
  89. reader_workbench/domains/plate_reader/plots/four_state_event_window/__init__.py +26 -0
  90. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic.py +295 -0
  91. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_components.py +149 -0
  92. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_render.py +308 -0
  93. reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_style.py +51 -0
  94. reader_workbench/domains/plate_reader/plots/four_state_event_window/schema.py +8 -0
  95. reader_workbench/domains/plate_reader/plots/four_state_event_window/summary.py +140 -0
  96. reader_workbench/domains/plate_reader/plots/grouping.py +53 -0
  97. reader_workbench/domains/plate_reader/plots/panels/__init__.py +12 -0
  98. reader_workbench/domains/plate_reader/plots/panels/snapshot.py +161 -0
  99. reader_workbench/domains/plate_reader/plots/panels/snapshot_data.py +91 -0
  100. reader_workbench/domains/plate_reader/plots/panels/time_series.py +296 -0
  101. reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic.py +435 -0
  102. reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic_render.py +300 -0
  103. reader_workbench/domains/plate_reader/plots/snapshot_barplot/__init__.py +315 -0
  104. reader_workbench/domains/plate_reader/plots/snapshot_barplot/planning.py +168 -0
  105. reader_workbench/domains/plate_reader/plots/snapshot_heatmap/__init__.py +205 -0
  106. reader_workbench/domains/plate_reader/plots/snapshot_heatmap/inputs.py +103 -0
  107. reader_workbench/domains/plate_reader/plots/time_series.py +317 -0
  108. reader_workbench/domains/plate_reader/plots/ts_and_snap/__init__.py +479 -0
  109. reader_workbench/domains/plate_reader/plots/ts_and_snap/planning.py +283 -0
  110. reader_workbench/domains/time_series/__init__.py +29 -0
  111. reader_workbench/domains/time_series/aggregation.py +60 -0
  112. reader_workbench/domains/time_series/contracts.py +368 -0
  113. reader_workbench/domains/time_series/reduction.py +395 -0
  114. reader_workbench/errors.py +55 -0
  115. reader_workbench/maintenance/__init__.py +6 -0
  116. reader_workbench/maintenance/docs.py +335 -0
  117. reader_workbench/maintenance/model.py +28 -0
  118. reader_workbench/maintenance/release.py +39 -0
  119. reader_workbench/maintenance/skills.py +124 -0
  120. reader_workbench/plotting/__init__.py +20 -0
  121. reader_workbench/plotting/mpl.py +56 -0
  122. reader_workbench/plotting/sinks.py +69 -0
  123. reader_workbench/plotting/style.py +175 -0
  124. reader_workbench/plotting/utils.py +27 -0
  125. reader_workbench/plugins/__init__.py +1 -0
  126. reader_workbench/plugins/catalog.py +33 -0
  127. reader_workbench/plugins/export/__init__.py +0 -0
  128. reader_workbench/plugins/export/_paths.py +21 -0
  129. reader_workbench/plugins/export/csv.py +41 -0
  130. reader_workbench/plugins/export/xlsx.py +44 -0
  131. reader_workbench/plugins/ingest/__init__.py +0 -0
  132. reader_workbench/plugins/ingest/_discovery.py +58 -0
  133. reader_workbench/plugins/ingest/discovery_policy.py +66 -0
  134. reader_workbench/plugins/ingest/flow_cytometer.py +139 -0
  135. reader_workbench/plugins/ingest/synergy_h1.py +234 -0
  136. reader_workbench/plugins/manifests/__init__.py +1 -0
  137. reader_workbench/plugins/manifests/export.py +29 -0
  138. reader_workbench/plugins/manifests/ingest.py +29 -0
  139. reader_workbench/plugins/manifests/plot.py +161 -0
  140. reader_workbench/plugins/manifests/transform.py +172 -0
  141. reader_workbench/plugins/manifests/validator.py +18 -0
  142. reader_workbench/plugins/plot/__init__.py +0 -0
  143. reader_workbench/plugins/plot/_shared.py +55 -0
  144. reader_workbench/plugins/plot/cytometry_diagnostic.py +54 -0
  145. reader_workbench/plugins/plot/distributions.py +58 -0
  146. reader_workbench/plugins/plot/dual_reporter_triptych.py +224 -0
  147. reader_workbench/plugins/plot/four_state_event_window_diagnostic.py +133 -0
  148. reader_workbench/plugins/plot/four_state_event_window_summary.py +53 -0
  149. reader_workbench/plugins/plot/four_state_vector_collection.py +45 -0
  150. reader_workbench/plugins/plot/four_state_vector_diagnostic.py +105 -0
  151. reader_workbench/plugins/plot/four_state_vector_heatmap.py +63 -0
  152. reader_workbench/plugins/plot/logic_symmetry.py +56 -0
  153. reader_workbench/plugins/plot/single_reporter_diagnostic.py +279 -0
  154. reader_workbench/plugins/plot/snapshot_barplot.py +65 -0
  155. reader_workbench/plugins/plot/snapshot_heatmap.py +104 -0
  156. reader_workbench/plugins/plot/time_series.py +114 -0
  157. reader_workbench/plugins/plot/ts_and_snap.py +210 -0
  158. reader_workbench/plugins/transform/__init__.py +0 -0
  159. reader_workbench/plugins/transform/_four_state_vector.py +204 -0
  160. reader_workbench/plugins/transform/_labeling.py +109 -0
  161. reader_workbench/plugins/transform/alias.py +70 -0
  162. reader_workbench/plugins/transform/assay_labels.py +62 -0
  163. reader_workbench/plugins/transform/blank.py +79 -0
  164. reader_workbench/plugins/transform/crosstalk_pairs.py +180 -0
  165. reader_workbench/plugins/transform/cytometry_gating.py +120 -0
  166. reader_workbench/plugins/transform/fold_change.py +79 -0
  167. reader_workbench/plugins/transform/four_state_event_window.py +93 -0
  168. reader_workbench/plugins/transform/four_state_vector.py +62 -0
  169. reader_workbench/plugins/transform/four_state_vector_collection.py +41 -0
  170. reader_workbench/plugins/transform/logic_symmetry.py +67 -0
  171. reader_workbench/plugins/transform/outlier_filter.py +60 -0
  172. reader_workbench/plugins/transform/overflow.py +197 -0
  173. reader_workbench/plugins/transform/ratio.py +237 -0
  174. reader_workbench/plugins/transform/sample_map.py +170 -0
  175. reader_workbench/plugins/transform/sample_metadata.py +94 -0
  176. reader_workbench/plugins/validator/__init__.py +1 -0
  177. reader_workbench/plugins/validator/to_tidy_plus_map.py +155 -0
  178. reader_workbench/protocols/__init__.py +80 -0
  179. reader_workbench/protocols/_builtins_plate_reader_growth.py +179 -0
  180. reader_workbench/protocols/_builtins_plate_reader_variants.py +274 -0
  181. reader_workbench/protocols/builtins.py +1656 -0
  182. reader_workbench/protocols/compiler.py +22 -0
  183. reader_workbench/protocols/compilers/__init__.py +1 -0
  184. reader_workbench/protocols/compilers/common.py +100 -0
  185. reader_workbench/protocols/compilers/cytometry.py +87 -0
  186. reader_workbench/protocols/compilers/generic.py +14 -0
  187. reader_workbench/protocols/compilers/logic.py +245 -0
  188. reader_workbench/protocols/compilers/plate_reader.py +937 -0
  189. reader_workbench/protocols/compilers/plate_reader_pipeline.py +197 -0
  190. reader_workbench/protocols/model.py +1486 -0
  191. reader_workbench/protocols/semantic_coverage.py +234 -0
  192. reader_workbench/runtime/__init__.py +12 -0
  193. reader_workbench/runtime/builtin.py +23 -0
  194. reader_workbench/runtime/model.py +42 -0
  195. reader_workbench/workbench/__init__.py +60 -0
  196. reader_workbench/workbench/assets/__init__.py +22 -0
  197. reader_workbench/workbench/assets/types.py +118 -0
  198. reader_workbench/workbench/audit/__init__.py +5 -0
  199. reader_workbench/workbench/audit/experiments.py +307 -0
  200. reader_workbench/workbench/audit/staging.py +187 -0
  201. reader_workbench/workbench/cli/__init__.py +51 -0
  202. reader_workbench/workbench/cli/_lazy.py +9 -0
  203. reader_workbench/workbench/cli/_records_view.py +150 -0
  204. reader_workbench/workbench/cli/_surface_execution.py +443 -0
  205. reader_workbench/workbench/cli/audit.py +95 -0
  206. reader_workbench/workbench/cli/automation.py +229 -0
  207. reader_workbench/workbench/cli/demo.py +46 -0
  208. reader_workbench/workbench/cli/dop.py +91 -0
  209. reader_workbench/workbench/cli/experiments.py +635 -0
  210. reader_workbench/workbench/cli/helpers.py +232 -0
  211. reader_workbench/workbench/cli/main.py +59 -0
  212. reader_workbench/workbench/cli/maintenance.py +82 -0
  213. reader_workbench/workbench/cli/notebooks.py +260 -0
  214. reader_workbench/workbench/cli/pagination.py +117 -0
  215. reader_workbench/workbench/cli/protocols.py +336 -0
  216. reader_workbench/workbench/cli/shared.py +309 -0
  217. reader_workbench/workbench/cli/surfaces.py +534 -0
  218. reader_workbench/workbench/cli/verification.py +128 -0
  219. reader_workbench/workbench/commands.py +10 -0
  220. reader_workbench/workbench/config/__init__.py +47 -0
  221. reader_workbench/workbench/config/identity.py +13 -0
  222. reader_workbench/workbench/config/load.py +405 -0
  223. reader_workbench/workbench/config/model.py +274 -0
  224. reader_workbench/workbench/context.py +26 -0
  225. reader_workbench/workbench/decl/__init__.py +31 -0
  226. reader_workbench/workbench/decl/build.py +190 -0
  227. reader_workbench/workbench/decl/model.py +81 -0
  228. reader_workbench/workbench/dop/__init__.py +12 -0
  229. reader_workbench/workbench/dop/builtins.py +261 -0
  230. reader_workbench/workbench/dop/model.py +209 -0
  231. reader_workbench/workbench/engine/__init__.py +42 -0
  232. reader_workbench/workbench/engine/_shared.py +76 -0
  233. reader_workbench/workbench/engine/contracts.py +283 -0
  234. reader_workbench/workbench/engine/execution.py +326 -0
  235. reader_workbench/workbench/engine/file_outputs.py +260 -0
  236. reader_workbench/workbench/engine/inputs.py +161 -0
  237. reader_workbench/workbench/engine/invocations.py +507 -0
  238. reader_workbench/workbench/engine/planning.py +72 -0
  239. reader_workbench/workbench/engine/runtime.py +464 -0
  240. reader_workbench/workbench/engine/setup.py +149 -0
  241. reader_workbench/workbench/engine/validation.py +684 -0
  242. reader_workbench/workbench/experiment/__init__.py +47 -0
  243. reader_workbench/workbench/experiment/model.py +381 -0
  244. reader_workbench/workbench/experiments.py +133 -0
  245. reader_workbench/workbench/graph/__init__.py +47 -0
  246. reader_workbench/workbench/graph/nodes.py +102 -0
  247. reader_workbench/workbench/graph/normalize.py +177 -0
  248. reader_workbench/workbench/graph/refs.py +148 -0
  249. reader_workbench/workbench/input_discovery.py +19 -0
  250. reader_workbench/workbench/inspection/__init__.py +3 -0
  251. reader_workbench/workbench/inspection/catalogs.py +128 -0
  252. reader_workbench/workbench/inspection/common.py +92 -0
  253. reader_workbench/workbench/inspection/dop.py +64 -0
  254. reader_workbench/workbench/inspection/experiments.py +449 -0
  255. reader_workbench/workbench/inspection/inventory.py +68 -0
  256. reader_workbench/workbench/inspection/protocols.py +368 -0
  257. reader_workbench/workbench/inspection/readiness.py +333 -0
  258. reader_workbench/workbench/inspection/reports.py +367 -0
  259. reader_workbench/workbench/inspection/results.py +166 -0
  260. reader_workbench/workbench/inspection/runtime.py +287 -0
  261. reader_workbench/workbench/inspection/semantics.py +192 -0
  262. reader_workbench/workbench/inspection/validation.py +30 -0
  263. reader_workbench/workbench/notebooks/__init__.py +17 -0
  264. reader_workbench/workbench/notebooks/_launch_registry.py +112 -0
  265. reader_workbench/workbench/notebooks/_launch_runtime.py +104 -0
  266. reader_workbench/workbench/notebooks/components/__init__.py +21 -0
  267. reader_workbench/workbench/notebooks/components/deliverables.py +403 -0
  268. reader_workbench/workbench/notebooks/components/overview.py +119 -0
  269. reader_workbench/workbench/notebooks/eda.marimo.py.txt +153 -0
  270. reader_workbench/workbench/notebooks/launch.py +274 -0
  271. reader_workbench/workbench/notebooks/presentation.py +136 -0
  272. reader_workbench/workbench/notebooks/scaffold.py +60 -0
  273. reader_workbench/workbench/ontology.py +78 -0
  274. reader_workbench/workbench/paths.py +44 -0
  275. reader_workbench/workbench/ports/__init__.py +31 -0
  276. reader_workbench/workbench/ports/model.py +168 -0
  277. reader_workbench/workbench/records/__init__.py +44 -0
  278. reader_workbench/workbench/records/epoch.py +329 -0
  279. reader_workbench/workbench/records/evidence.py +247 -0
  280. reader_workbench/workbench/records/identity.py +87 -0
  281. reader_workbench/workbench/records/locking.py +185 -0
  282. reader_workbench/workbench/records/model.py +711 -0
  283. reader_workbench/workbench/records/sources.py +73 -0
  284. reader_workbench/workbench/records/store.py +1022 -0
  285. reader_workbench/workbench/records/verification.py +998 -0
  286. reader_workbench/workbench/registry.py +333 -0
  287. reader_workbench/workbench/spec_overrides.py +215 -0
  288. reader_workbench-1.0.0.dist-info/METADATA +91 -0
  289. reader_workbench-1.0.0.dist-info/RECORD +293 -0
  290. reader_workbench-1.0.0.dist-info/WHEEL +5 -0
  291. reader_workbench-1.0.0.dist-info/entry_points.txt +2 -0
  292. reader_workbench-1.0.0.dist-info/licenses/LICENSE +21 -0
  293. reader_workbench-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,141 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ from collections.abc import Mapping, Sequence
5
+
6
+ import pandas as pd
7
+
8
+ from ._shared import (
9
+ canonical_channel,
10
+ coerce_measurements,
11
+ is_blank_measurement,
12
+ require,
13
+ resolve_channel,
14
+ time_column,
15
+ well_headers,
16
+ )
17
+
18
+
19
+ def tidy_kinetic_blocks(
20
+ kinetic: pd.DataFrame,
21
+ *,
22
+ elapsed_h: float,
23
+ sheet_index: int,
24
+ sheet_name: str,
25
+ channels: Sequence[str] | None,
26
+ channel_map_ci: Mapping[str, str],
27
+ map_free_raw_by_channel: dict[str, tuple[str, str, str]],
28
+ ) -> pd.DataFrame:
29
+ logger = logging.getLogger("reader")
30
+ raw_to_resolved: dict[str, str] = {}
31
+ raw_to_canonical: dict[str, str] = {}
32
+
33
+ def row_has(frame: pd.DataFrame, index: int, pattern: str) -> bool:
34
+ return frame.iloc[index].astype(str).str.contains(pattern, case=False, na=False).any()
35
+
36
+ def looks_like_label(first_cell: object) -> bool:
37
+ value = str(first_cell or "").strip()
38
+ if not value or value.lower() == "nan":
39
+ return False
40
+ if ":" in value:
41
+ return True
42
+ try:
43
+ return resolve_channel(value, channels=channels, channel_map_ci=channel_map_ci) is not None
44
+ except ValueError as error:
45
+ raise ValueError(f"{error} in kinetic sheet {sheet_name!r}") from error
46
+
47
+ label_rows = [index for index, row in kinetic.iterrows() if looks_like_label(row.iat[0])]
48
+ require(
49
+ label_rows,
50
+ "No kinetic blocks found: expected a channel label in column A "
51
+ "(e.g., 'OD600' or 'OD600: …') preceding a 'Time' header row.",
52
+ )
53
+
54
+ parts: list[pd.DataFrame] = []
55
+ for block_index, start in enumerate(label_rows):
56
+ end = label_rows[block_index + 1] if block_index + 1 < len(label_rows) else len(kinetic)
57
+ block = kinetic.iloc[start:end].reset_index(drop=True)
58
+ section_end = next((index for index in range(1, len(block)) if row_has(block, index, r"^Results$")), None)
59
+ if section_end is not None:
60
+ block = block.iloc[:section_end].reset_index(drop=True)
61
+
62
+ raw_channel = str(block.iat[0, 0]).strip()
63
+ canonical = canonical_channel(raw_channel)
64
+ channel = resolve_channel(raw_channel, channels=channels, channel_map_ci=channel_map_ci)
65
+ if channel is None:
66
+ continue
67
+
68
+ if not channel_map_ci:
69
+ normalized_canonical = canonical.lower()
70
+ previous_raw = map_free_raw_by_channel.get(channel)
71
+ if previous_raw is not None and previous_raw[0] != normalized_canonical:
72
+ raise ValueError(
73
+ f"Ambiguous map-free kinetic channel {channel!r} in sheet {sheet_name!r}: "
74
+ f"raw labels {previous_raw[1]!r} from sheet {previous_raw[2]!r} and "
75
+ f"{canonical!r} from sheet {sheet_name!r} resolve to the same declared channel; "
76
+ "provide channel_map to select raw labels explicitly"
77
+ )
78
+ if previous_raw is None:
79
+ map_free_raw_by_channel[channel] = (normalized_canonical, canonical, sheet_name)
80
+
81
+ raw_to_canonical.setdefault(raw_channel, canonical)
82
+ previous = raw_to_resolved.get(raw_channel)
83
+ if previous is not None and previous != channel:
84
+ logger.warning(
85
+ "[warn]inconsistent channel resolution[/warn] • raw=%r canon=%r previously→%r now→%r",
86
+ raw_channel,
87
+ canonical,
88
+ previous,
89
+ channel,
90
+ )
91
+ raw_to_resolved[raw_channel] = channel
92
+
93
+ header_index = next((index for index in range(1, len(block)) if row_has(block, index, r"^Time")), None)
94
+ require(header_index is not None, f"Time header not found in kinetic block for channel {channel!r}")
95
+ header = block.iloc[header_index].astype(str).tolist()
96
+ time_key = time_column(header)
97
+ wells = well_headers(header)
98
+ require(wells, f"No well columns (A1..H12) in kinetic header for {channel!r}")
99
+
100
+ data = block.iloc[header_index + 1 :].reset_index(drop=True)
101
+ data.columns = header
102
+ raw_times = data[time_key]
103
+ parsed_times = pd.to_timedelta(raw_times, errors="coerce")
104
+ has_measurement = data[wells].map(lambda value: not is_blank_measurement(value)).any(axis=1)
105
+ invalid_time = parsed_times.isna() & (
106
+ raw_times.map(lambda value: not is_blank_measurement(value)) | has_measurement
107
+ )
108
+ if invalid_time.any():
109
+ index = invalid_time[invalid_time].index[0]
110
+ token = raw_times.loc[index]
111
+ raise ValueError(f"Invalid kinetic time token {token!r} in sheet {sheet_name!r}, channel {channel!r}")
112
+ time_hours = parsed_times.dt.total_seconds() / 3600.0
113
+ data = data.assign(__time_hr=time_hours).loc[lambda frame: frame["__time_hr"].notna()].reset_index(drop=True)
114
+ require(not data.empty, f"Non-parsable time values in kinetic block for {channel!r}")
115
+
116
+ first_relative_time = float(data["__time_hr"].iloc[0])
117
+ data["__time_hr"] = float(elapsed_h) + (data["__time_hr"] - first_relative_time)
118
+
119
+ melted = data.melt(
120
+ id_vars=["__time_hr"],
121
+ value_vars=wells,
122
+ var_name="position",
123
+ value_name="value",
124
+ ).rename(columns={"__time_hr": "time"})
125
+ melted["channel"] = channel
126
+ melted["sheet_index"] = sheet_index
127
+ melted["sheet_name"] = sheet_name
128
+ melted["source"] = "kinetic"
129
+ parts.append(melted)
130
+
131
+ require(parts, f"No configured kinetic channel blocks found in sheet {sheet_name!r}")
132
+ result = coerce_measurements(pd.concat(parts, ignore_index=True), source="kinetic")
133
+
134
+ if raw_to_resolved:
135
+ pairs = [
136
+ f"{raw!r} → {raw_to_canonical.get(raw, '?')!r} → {raw_to_resolved[raw]!r}"
137
+ for raw in sorted(raw_to_resolved)
138
+ ]
139
+ logger.debug("channel normalization (kinetic, sheet %s): %s", sheet_name, "; ".join(pairs))
140
+
141
+ return result
@@ -0,0 +1,295 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ import math
5
+ from collections.abc import Mapping, Sequence
6
+ from pathlib import Path
7
+
8
+ import numpy as np
9
+ import pandas as pd
10
+
11
+ from ._kinetic import tidy_kinetic_blocks
12
+ from ._shared import (
13
+ ensure_excel_path,
14
+ extract_sheet_datetime,
15
+ is_blank_measurement,
16
+ normalize_channel_map,
17
+ normalize_time_series,
18
+ require,
19
+ require_unique_measurements,
20
+ resolve_channel,
21
+ )
22
+ from ._snapshot import tidy_snapshot_block
23
+
24
+
25
+ def _selected_sheets(excel: pd.ExcelFile, sheet_names: Sequence[str] | None) -> list[str]:
26
+ sheets = list(sheet_names or excel.sheet_names)
27
+ require(sheets, "Workbook has no sheets")
28
+ for sheet in sheets:
29
+ require(sheet in excel.sheet_names, f"Sheet {sheet!r} not found in workbook")
30
+ return sheets
31
+
32
+
33
+ def _elapsed_hours_by_sheet(excel: pd.ExcelFile, sheets: Sequence[str]) -> dict[str, float]:
34
+ datetimes = {sheet: extract_sheet_datetime(excel, sheet) for sheet in sheets}
35
+ first_datetime = min(datetimes.values())
36
+ return {sheet: (value - first_datetime).total_seconds() / 3600.0 for sheet, value in datetimes.items()}
37
+
38
+
39
+ def _split_snapshot_vs_kinetic(
40
+ frame: pd.DataFrame,
41
+ *,
42
+ channels: Sequence[str] | None,
43
+ channel_map_ci: Mapping[str, str],
44
+ ) -> tuple[pd.DataFrame | None, pd.DataFrame | None]:
45
+ def row_has(index: int, keyword: str) -> bool:
46
+ return frame.iloc[index].astype(str).str.contains(keyword, case=False, na=False).any()
47
+
48
+ results_index = next((index for index in frame.index if row_has(index, "Results")), None)
49
+ if results_index is None:
50
+ return None, None
51
+
52
+ start = results_index + 1
53
+ for index in frame.index[start:]:
54
+ if frame.iloc[index].dropna(how="all").empty:
55
+ start = index + 1
56
+ break
57
+ time_index = next((index for index in frame.index[start:] if row_has(index, "Time")), None)
58
+
59
+ def starts_kinetic_block(index: int) -> bool:
60
+ first_cell = frame.iat[index, 0]
61
+ if is_blank_measurement(first_cell):
62
+ return False
63
+ return resolve_channel(str(first_cell), channels=channels, channel_map_ci=channel_map_ci) is not None
64
+
65
+ channel_index = next((index for index in frame.index[start:] if starts_kinetic_block(index)), None)
66
+ boundaries = [index for index in (time_index, channel_index) if index is not None]
67
+ cut = min(boundaries) if boundaries else None
68
+ snapshot = (
69
+ frame.iloc[start:cut].reset_index(drop=True) if cut is not None else frame.iloc[start:].reset_index(drop=True)
70
+ )
71
+ kinetic = frame.iloc[cut:].reset_index(drop=True) if cut is not None else None
72
+ return snapshot, kinetic
73
+
74
+
75
+ def _find_kinetic_section(
76
+ frame: pd.DataFrame,
77
+ *,
78
+ sheet_name: str,
79
+ channels: Sequence[str] | None,
80
+ channel_map_ci: Mapping[str, str],
81
+ ) -> pd.DataFrame | None:
82
+ for index in frame.index:
83
+ cell = str(frame.iat[index, 0]).strip()
84
+ if not cell:
85
+ continue
86
+ try:
87
+ resolved = resolve_channel(cell, channels=channels, channel_map_ci=channel_map_ci)
88
+ except ValueError as error:
89
+ raise ValueError(f"{error} in kinetic sheet {sheet_name!r}") from error
90
+ if resolved is not None:
91
+ return frame.iloc[index:].reset_index(drop=True)
92
+ return None
93
+
94
+
95
+ def _finalize_measurements(
96
+ frame: pd.DataFrame,
97
+ *,
98
+ channels: Sequence[str] | None,
99
+ expected_channels: set[str],
100
+ missing_message: str,
101
+ time_round_decimals: int | None,
102
+ time_step_h: float | None,
103
+ time_offset_h: float,
104
+ filter_to_channels: bool,
105
+ ) -> pd.DataFrame:
106
+ require(not isinstance(time_offset_h, (bool, np.bool_)), "time_offset_h must not be a boolean")
107
+ require(math.isfinite(time_offset_h), "time_offset_h must be finite")
108
+ require(time_offset_h >= 0.0, "time_offset_h must be greater than or equal to 0")
109
+ result = frame
110
+ if filter_to_channels and channels:
111
+ result = result[result["channel"].isin(channels)].reset_index(drop=True)
112
+
113
+ result["time"] = normalize_time_series(
114
+ result["time"],
115
+ time_round_decimals=time_round_decimals,
116
+ time_step_h=time_step_h,
117
+ )
118
+ with np.errstate(over="ignore", invalid="ignore"):
119
+ result["time"] = result["time"] + float(time_offset_h)
120
+ require(
121
+ result["time"].map(math.isfinite).all(),
122
+ "time values must remain finite after applying time_offset_h",
123
+ )
124
+ if time_round_decimals is not None:
125
+ with np.errstate(over="ignore", invalid="ignore"):
126
+ result["time"] = result["time"].round(int(time_round_decimals))
127
+ require(
128
+ result["time"].map(math.isfinite).all(),
129
+ "time values must remain finite after applying time_offset_h",
130
+ )
131
+ result["value"] = pd.to_numeric(result["value"], errors="raise")
132
+ require(result["time"].ge(0).all(), "Internal error: negative time encountered after alignment")
133
+ require(result["time"].notna().all(), "Internal error: time contains NaN after alignment")
134
+
135
+ missing = expected_channels - set(result["channel"].astype(str).unique())
136
+ require(not missing, f"{missing_message}: {sorted(missing)}")
137
+
138
+ result["position"] = result["position"].astype(str)
139
+ result["channel"] = result["channel"].astype(str)
140
+ require_unique_measurements(result)
141
+ return result.reset_index(drop=True)
142
+
143
+
144
+ def parse_snapshot_and_timeseries(
145
+ path: str | Path,
146
+ *,
147
+ channels: Sequence[str] | None = None,
148
+ channel_map: Mapping[str, str] | None = None,
149
+ sheet_names: Sequence[str] | None = None,
150
+ time_round_decimals: int | None = 12,
151
+ time_step_h: float | None = None,
152
+ time_offset_h: float = 0.0,
153
+ include_snapshot: bool = True,
154
+ include_kinetic: bool = True,
155
+ ) -> pd.DataFrame:
156
+ """Parse snapshot and kinetic blocks from one Synergy H1 workbook."""
157
+ workbook = Path(path)
158
+ ensure_excel_path(workbook)
159
+ channel_map_ci = normalize_channel_map(channel_map)
160
+ require(channels or channel_map_ci, "Provide either 'channels' or 'channel_map'")
161
+ expected_channels = set(channels or channel_map_ci.values())
162
+
163
+ frames: list[pd.DataFrame] = []
164
+ map_free_raw_by_channel: dict[str, tuple[str, str, str]] = {}
165
+ with pd.ExcelFile(workbook) as excel:
166
+ sheets = _selected_sheets(excel, sheet_names)
167
+ elapsed_by_sheet = _elapsed_hours_by_sheet(excel, sheets)
168
+ for sheet_index, sheet in enumerate(sheets):
169
+ raw = excel.parse(sheet_name=sheet, header=None, dtype=str)
170
+ snapshot, kinetic = _split_snapshot_vs_kinetic(
171
+ raw,
172
+ channels=channels,
173
+ channel_map_ci=channel_map_ci,
174
+ )
175
+
176
+ if include_snapshot and snapshot is not None and not snapshot.empty:
177
+ frames.append(
178
+ tidy_snapshot_block(
179
+ snapshot,
180
+ elapsed_h=elapsed_by_sheet[sheet],
181
+ sheet_index=sheet_index,
182
+ sheet_name=sheet,
183
+ channels=channels,
184
+ channel_map_ci=channel_map_ci,
185
+ )
186
+ )
187
+ declared_channels = sorted(set(channels or channel_map_ci.values()))
188
+ logging.getLogger("reader").debug(
189
+ "channel normalization (snapshot, sheet %s): declared=%s",
190
+ sheet,
191
+ declared_channels,
192
+ )
193
+ if include_kinetic and kinetic is not None and not kinetic.empty:
194
+ frames.append(
195
+ tidy_kinetic_blocks(
196
+ kinetic,
197
+ elapsed_h=elapsed_by_sheet[sheet],
198
+ sheet_index=sheet_index,
199
+ sheet_name=sheet,
200
+ channels=channels,
201
+ channel_map_ci=channel_map_ci,
202
+ map_free_raw_by_channel=map_free_raw_by_channel,
203
+ )
204
+ )
205
+
206
+ require(frames, f"No parsable data found in {workbook.name}")
207
+ result = pd.concat(frames, ignore_index=True)
208
+
209
+ requested_sources = {
210
+ source for source, requested in (("snapshot", include_snapshot), ("kinetic", include_kinetic)) if requested
211
+ }
212
+ sheet_has_snapshot = result.groupby("sheet_index")["source"].transform(lambda values: (values == "snapshot").any())
213
+ sheet_min_time = result.groupby("sheet_index")["time"].transform("min")
214
+ overlapping_initial_kinetic = (
215
+ (result["source"] == "kinetic") & sheet_has_snapshot & (result["time"] == sheet_min_time)
216
+ )
217
+ result = result.loc[~overlapping_initial_kinetic]
218
+
219
+ observed_sources = set(result["source"].astype(str).unique())
220
+ missing_sources = requested_sources - observed_sources
221
+ require(not missing_sources, f"Missing requested Synergy data sources: {sorted(missing_sources)}")
222
+ for source in sorted(requested_sources):
223
+ observed_channels = set(result.loc[result["source"] == source, "channel"].astype(str).unique())
224
+ missing_source_channels = expected_channels - observed_channels
225
+ require(
226
+ not missing_source_channels,
227
+ f"{source.title()} data missing for channels: {sorted(missing_source_channels)}",
228
+ )
229
+
230
+ return _finalize_measurements(
231
+ result,
232
+ channels=channels,
233
+ expected_channels=expected_channels,
234
+ missing_message="Missing data for channels",
235
+ time_round_decimals=time_round_decimals,
236
+ time_step_h=time_step_h,
237
+ time_offset_h=time_offset_h,
238
+ filter_to_channels=True,
239
+ )
240
+
241
+
242
+ def parse_kinetic_only(
243
+ path: str | Path,
244
+ *,
245
+ channels: Sequence[str] | None = None,
246
+ channel_map: Mapping[str, str] | None = None,
247
+ sheet_names: Sequence[str] | None = None,
248
+ time_round_decimals: int | None = 12,
249
+ time_step_h: float | None = None,
250
+ time_offset_h: float = 0.0,
251
+ ) -> pd.DataFrame:
252
+ """Parse kinetic blocks from one Synergy H1 workbook."""
253
+ workbook = Path(path)
254
+ ensure_excel_path(workbook)
255
+ channel_map_ci = normalize_channel_map(channel_map)
256
+ require(channels or channel_map_ci, "Provide either 'channels' or 'channel_map'")
257
+
258
+ frames: list[pd.DataFrame] = []
259
+ map_free_raw_by_channel: dict[str, tuple[str, str, str]] = {}
260
+ with pd.ExcelFile(workbook) as excel:
261
+ sheets = _selected_sheets(excel, sheet_names)
262
+ elapsed_by_sheet = _elapsed_hours_by_sheet(excel, sheets)
263
+ for sheet_index, sheet in enumerate(sheets):
264
+ raw = excel.parse(sheet_name=sheet, header=None, dtype=str)
265
+ kinetic = _find_kinetic_section(
266
+ raw,
267
+ sheet_name=sheet,
268
+ channels=channels,
269
+ channel_map_ci=channel_map_ci,
270
+ )
271
+ require(kinetic is not None, f"No kinetic data found in sheet {sheet!r}")
272
+ frames.append(
273
+ tidy_kinetic_blocks(
274
+ kinetic,
275
+ elapsed_h=elapsed_by_sheet[sheet],
276
+ sheet_index=sheet_index,
277
+ sheet_name=sheet,
278
+ channels=channels,
279
+ channel_map_ci=channel_map_ci,
280
+ map_free_raw_by_channel=map_free_raw_by_channel,
281
+ )
282
+ )
283
+
284
+ require(frames, f"No kinetic readings found in {workbook.name}")
285
+ expected_channels = {str(channel) for channel in (channels or channel_map_ci.values())}
286
+ return _finalize_measurements(
287
+ pd.concat(frames, ignore_index=True),
288
+ channels=channels,
289
+ expected_channels=expected_channels,
290
+ missing_message="Kinetic data missing for channels",
291
+ time_round_decimals=time_round_decimals,
292
+ time_step_h=time_step_h,
293
+ time_offset_h=time_offset_h,
294
+ filter_to_channels=False,
295
+ )
@@ -0,0 +1,195 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from collections.abc import Iterable, Mapping, Sequence
5
+ from datetime import datetime
6
+ from pathlib import Path
7
+
8
+ import numpy as np
9
+ import pandas as pd
10
+
11
+
12
+ def require(condition: bool, message: str) -> None:
13
+ if not condition:
14
+ raise ValueError(message)
15
+
16
+
17
+ def drop_all_empty_rows(frame: pd.DataFrame) -> pd.DataFrame:
18
+ with pd.option_context("future.no_silent_downcasting", True):
19
+ normalized = frame.replace(r"^\s*$", pd.NA, regex=True)
20
+ return normalized.infer_objects(copy=False).dropna(how="all").reset_index(drop=True)
21
+
22
+
23
+ def ensure_excel_path(path: Path) -> None:
24
+ require(path.exists(), f"Input file not found: {path}")
25
+ require(path.is_file(), f"Synergy H1 input is not a regular file: {path}")
26
+ require(
27
+ path.suffix.lower() == ".xlsx",
28
+ f"Synergy H1 ingest requires a modern .xlsx workbook, got {path.suffix or '<no extension>'!r}",
29
+ )
30
+
31
+
32
+ def require_unique_measurements(frame: pd.DataFrame) -> None:
33
+ keys = ["position", "channel", "time"]
34
+ duplicate_mask = frame.duplicated(subset=keys, keep=False)
35
+ if not duplicate_mask.any():
36
+ return
37
+ preview = frame.loc[duplicate_mask, keys].drop_duplicates().head(8).to_dict(orient="records")
38
+ raise ValueError(f"Synergy measurements must be unique by {keys}; duplicate keys include {preview}")
39
+
40
+
41
+ def probe_synergy_workbook(path: str | Path) -> tuple[str, ...]:
42
+ """Open a Synergy workbook and return its sheet names without parsing sheet data."""
43
+ workbook = Path(path)
44
+ ensure_excel_path(workbook)
45
+ with pd.ExcelFile(workbook) as excel:
46
+ return tuple(str(sheet) for sheet in excel.sheet_names)
47
+
48
+
49
+ def extract_sheet_datetime(excel: pd.ExcelFile, sheet: str) -> datetime:
50
+ metadata = excel.parse(sheet_name=sheet, header=None, nrows=20, dtype=str)
51
+
52
+ def row_has(index: int, pattern: str) -> bool:
53
+ return metadata.iloc[index].astype(str).str.fullmatch(pattern, case=False).any()
54
+
55
+ date_row = next((index for index in metadata.index if row_has(index, r"Date")), None)
56
+ time_row = next((index for index in metadata.index if row_has(index, r"Time")), None)
57
+ require(date_row is not None and time_row is not None, f"Missing 'Date'/'Time' rows in sheet {sheet!r}")
58
+ date = pd.to_datetime(metadata.iloc[date_row, 1]).date()
59
+ time = pd.to_datetime(metadata.iloc[time_row, 1]).time()
60
+ return datetime.combine(date, time)
61
+
62
+
63
+ def normalize_time_series(
64
+ time: pd.Series,
65
+ *,
66
+ time_round_decimals: int | None,
67
+ time_step_h: float | None,
68
+ ) -> pd.Series:
69
+ normalized = pd.to_numeric(time, errors="raise")
70
+ if time_step_h is not None:
71
+ step = float(time_step_h)
72
+ require(step > 0, "time_step_h must be > 0")
73
+ normalized = (normalized / step).round() * step
74
+ if time_round_decimals is not None:
75
+ decimals = int(time_round_decimals)
76
+ require(decimals >= 0, "time_round_decimals must be >= 0")
77
+ normalized = normalized.round(decimals)
78
+ return normalized
79
+
80
+
81
+ def canonical_channel(value: str) -> str:
82
+ raw = str(value or "").strip()
83
+ prefix, separator, suffix = raw.partition(":")
84
+ normalized_prefix = re.sub(r"\s+[AB]$", "", prefix.strip(), flags=re.IGNORECASE)
85
+ normalized_prefix = re.sub(r"\s+", " ", normalized_prefix).strip()
86
+ if not separator:
87
+ return normalized_prefix
88
+ normalized_suffix = re.sub(r"\s+", "", suffix)
89
+ return f"{normalized_prefix}:{normalized_suffix}"
90
+
91
+
92
+ _OVERFLOW_TOKENS = frozenset({"overflow", "ovrflw", "ovr", "over", "inf", "infinity", "∞"})
93
+ _GREATER_THAN_NUMBER = re.compile(r"^>\s*[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?$")
94
+
95
+
96
+ def is_overflow_token(value: object) -> bool:
97
+ normalized = str(value or "").strip()
98
+ if not normalized:
99
+ return False
100
+ return normalized.lower() in _OVERFLOW_TOKENS or bool(_GREATER_THAN_NUMBER.fullmatch(normalized))
101
+
102
+
103
+ def normalize_channel_map(channel_map: Mapping[str, str] | None) -> Mapping[str, str]:
104
+ if not channel_map:
105
+ return {}
106
+ normalized: dict[str, str] = {}
107
+ for raw_label, output_channel in channel_map.items():
108
+ key = canonical_channel(str(raw_label)).lower()
109
+ value = str(output_channel).strip()
110
+ require(key != "", "channel_map keys must contain a channel label")
111
+ require(value != "", f"channel_map[{raw_label!r}] must name an output channel")
112
+ require(
113
+ key not in normalized,
114
+ f"Duplicate channel_map declaration after normalization: {raw_label!r} resolves to {key!r}",
115
+ )
116
+ normalized[key] = value
117
+ return normalized
118
+
119
+
120
+ def resolve_channel_from_map(raw_label: str, *, channel_map_ci: Mapping[str, str]) -> str | None:
121
+ label = canonical_channel(raw_label).lower()
122
+ return channel_map_ci.get(label)
123
+
124
+
125
+ def resolve_channel(
126
+ raw_label: str,
127
+ *,
128
+ channels: Sequence[str] | None,
129
+ channel_map_ci: Mapping[str, str],
130
+ ) -> str | None:
131
+ normalized_label = canonical_channel(raw_label).lower()
132
+ mapped = resolve_channel_from_map(raw_label, channel_map_ci=channel_map_ci)
133
+ if mapped is not None:
134
+ return mapped
135
+
136
+ if channels:
137
+ declared = [(canonical_channel(channel).lower(), channel) for channel in channels]
138
+ matches = [channel for normalized, channel in declared if normalized == normalized_label]
139
+ if not matches and not channel_map_ci and ":" in normalized_label:
140
+ base_label = normalized_label.partition(":")[0]
141
+ matches = [channel for normalized, channel in declared if normalized == base_label]
142
+ require(
143
+ len(matches) <= 1,
144
+ f"Ambiguous configured channels for raw label {raw_label!r}: {matches}",
145
+ )
146
+ return matches[0] if matches else None
147
+
148
+ if channel_map_ci:
149
+ return None
150
+ raise ValueError("Provide at least one of: channels or channel_map")
151
+
152
+
153
+ def time_column(header: Iterable[str]) -> str:
154
+ for value in header:
155
+ if str(value).strip().lower().startswith("time"):
156
+ return str(value)
157
+ raise ValueError("Time column not found in kinetic block header")
158
+
159
+
160
+ def well_headers(header: Iterable[str]) -> list[str]:
161
+ candidates = [str(value).strip() for value in header if re.fullmatch(r"[A-H][0-9]+", str(value).strip())]
162
+ invalid = [value for value in candidates if not re.fullmatch(r"[A-H](?:[1-9]|1[0-2])", value)]
163
+ require(not invalid, f"Kinetic header contains positions outside the 96-well range A1..H12: {invalid}")
164
+ duplicates = sorted({value for value in candidates if candidates.count(value) > 1})
165
+ require(not duplicates, f"Kinetic header contains duplicate well columns: {duplicates}")
166
+ return candidates
167
+
168
+
169
+ def is_blank_measurement(value: object) -> bool:
170
+ return value is None or bool(pd.isna(value)) or str(value).strip() == ""
171
+
172
+
173
+ def coerce_measurements(frame: pd.DataFrame, *, source: str) -> pd.DataFrame:
174
+ result = frame.loc[~frame["value"].map(is_blank_measurement)].copy()
175
+ if result.empty:
176
+ result["overflow"] = pd.Series(dtype=bool)
177
+ result["value"] = pd.Series(dtype=float)
178
+ return result
179
+
180
+ raw_values = result["value"].astype(str)
181
+ overflow = raw_values.map(is_overflow_token)
182
+ numeric = pd.to_numeric(raw_values, errors="coerce")
183
+ invalid = numeric.isna() & ~overflow
184
+ if invalid.any():
185
+ index = invalid[invalid].index[0]
186
+ row = result.loc[index]
187
+ raise ValueError(
188
+ f"Invalid {source} measurement token {raw_values.loc[index]!r} in sheet {row['sheet_name']!r}, "
189
+ f"channel {row['channel']!r}, well {row['position']!r}"
190
+ )
191
+
192
+ result["overflow"] = overflow.astype(bool)
193
+ result["value"] = numeric.astype(float)
194
+ result.loc[overflow, "value"] = np.inf
195
+ return result.reset_index(drop=True)