reader-workbench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- reader_workbench/__init__.py +22 -0
- reader_workbench/__main__.py +4 -0
- reader_workbench/_version.py +17 -0
- reader_workbench/api/__init__.py +74 -0
- reader_workbench/api/_record_reads.py +75 -0
- reader_workbench/api/artifacts.py +79 -0
- reader_workbench/api/facade.py +538 -0
- reader_workbench/api/models.py +285 -0
- reader_workbench/api/notebooks.py +63 -0
- reader_workbench/contracts/__init__.py +18 -0
- reader_workbench/contracts/builtins/__init__.py +36 -0
- reader_workbench/contracts/builtins/cytometry.py +140 -0
- reader_workbench/contracts/builtins/four_state_event_window.py +214 -0
- reader_workbench/contracts/builtins/generic.py +21 -0
- reader_workbench/contracts/builtins/logic.py +149 -0
- reader_workbench/contracts/builtins/plate_reader.py +47 -0
- reader_workbench/contracts/catalog.py +257 -0
- reader_workbench/contracts/model.py +109 -0
- reader_workbench/domains/__init__.py +1 -0
- reader_workbench/domains/cytometry/__init__.py +3 -0
- reader_workbench/domains/cytometry/analysis/__init__.py +29 -0
- reader_workbench/domains/cytometry/analysis/events.py +182 -0
- reader_workbench/domains/cytometry/analysis/gating.py +175 -0
- reader_workbench/domains/cytometry/analysis/workflow.py +248 -0
- reader_workbench/domains/cytometry/io/__init__.py +3 -0
- reader_workbench/domains/cytometry/io/fcs.py +135 -0
- reader_workbench/domains/cytometry/plots/__init__.py +5 -0
- reader_workbench/domains/cytometry/plots/diagnostic.py +155 -0
- reader_workbench/domains/logic/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/pairs.py +661 -0
- reader_workbench/domains/logic/four_state_vector/__init__.py +7 -0
- reader_workbench/domains/logic/four_state_vector/builder.py +321 -0
- reader_workbench/domains/logic/four_state_vector/collection/__init__.py +20 -0
- reader_workbench/domains/logic/four_state_vector/collection/checks.py +49 -0
- reader_workbench/domains/logic/four_state_vector/collection/constants.py +29 -0
- reader_workbench/domains/logic/four_state_vector/collection/model.py +19 -0
- reader_workbench/domains/logic/four_state_vector/collection/render.py +383 -0
- reader_workbench/domains/logic/four_state_vector/collection/sources.py +185 -0
- reader_workbench/domains/logic/four_state_vector/config.py +214 -0
- reader_workbench/domains/logic/four_state_vector/diagnostic.py +361 -0
- reader_workbench/domains/logic/four_state_vector/heatmap.py +86 -0
- reader_workbench/domains/logic/four_state_vector/math.py +191 -0
- reader_workbench/domains/logic/four_state_vector/reference.py +85 -0
- reader_workbench/domains/logic/four_state_vector/selection.py +228 -0
- reader_workbench/domains/logic/four_state_vector/treatment_semantics.py +51 -0
- reader_workbench/domains/logic/four_state_vector/validation.py +19 -0
- reader_workbench/domains/logic/logic_symmetry/__init__.py +3 -0
- reader_workbench/domains/logic/logic_symmetry/encodings.py +93 -0
- reader_workbench/domains/logic/logic_symmetry/extract_corners.py +192 -0
- reader_workbench/domains/logic/logic_symmetry/main.py +236 -0
- reader_workbench/domains/logic/logic_symmetry/metrics.py +96 -0
- reader_workbench/domains/logic/logic_symmetry/overlay.py +129 -0
- reader_workbench/domains/logic/logic_symmetry/prep.py +138 -0
- reader_workbench/domains/logic/logic_symmetry/render.py +356 -0
- reader_workbench/domains/logic/treatment_columns.py +42 -0
- reader_workbench/domains/plate_reader/__init__.py +1 -0
- reader_workbench/domains/plate_reader/analysis/__init__.py +14 -0
- reader_workbench/domains/plate_reader/analysis/fold_change.py +474 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/__init__.py +21 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/aggregation.py +191 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contract_fields.py +50 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contracts.py +320 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/design_dispositions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/disposition_records.py +140 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/event_sensitivity.py +27 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/materialize.py +274 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/observation_resampling.py +97 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/reduction.py +62 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/seeds.py +15 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/sources.py +308 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusion_validation.py +94 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/timepoints.py +76 -0
- reader_workbench/domains/plate_reader/io/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/sample_map.py +65 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_kinetic.py +141 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_parser.py +295 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_shared.py +195 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_snapshot.py +130 -0
- reader_workbench/domains/plate_reader/ordering.py +59 -0
- reader_workbench/domains/plate_reader/plots/__init__.py +15 -0
- reader_workbench/domains/plate_reader/plots/_data.py +29 -0
- reader_workbench/domains/plate_reader/plots/common.py +346 -0
- reader_workbench/domains/plate_reader/plots/distributions.py +324 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych.py +525 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych_render.py +195 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/__init__.py +26 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic.py +295 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_components.py +149 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_render.py +308 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_style.py +51 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/schema.py +8 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/summary.py +140 -0
- reader_workbench/domains/plate_reader/plots/grouping.py +53 -0
- reader_workbench/domains/plate_reader/plots/panels/__init__.py +12 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot.py +161 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot_data.py +91 -0
- reader_workbench/domains/plate_reader/plots/panels/time_series.py +296 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic.py +435 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic_render.py +300 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/__init__.py +315 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/planning.py +168 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/__init__.py +205 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/inputs.py +103 -0
- reader_workbench/domains/plate_reader/plots/time_series.py +317 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/__init__.py +479 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/planning.py +283 -0
- reader_workbench/domains/time_series/__init__.py +29 -0
- reader_workbench/domains/time_series/aggregation.py +60 -0
- reader_workbench/domains/time_series/contracts.py +368 -0
- reader_workbench/domains/time_series/reduction.py +395 -0
- reader_workbench/errors.py +55 -0
- reader_workbench/maintenance/__init__.py +6 -0
- reader_workbench/maintenance/docs.py +335 -0
- reader_workbench/maintenance/model.py +28 -0
- reader_workbench/maintenance/release.py +39 -0
- reader_workbench/maintenance/skills.py +124 -0
- reader_workbench/plotting/__init__.py +20 -0
- reader_workbench/plotting/mpl.py +56 -0
- reader_workbench/plotting/sinks.py +69 -0
- reader_workbench/plotting/style.py +175 -0
- reader_workbench/plotting/utils.py +27 -0
- reader_workbench/plugins/__init__.py +1 -0
- reader_workbench/plugins/catalog.py +33 -0
- reader_workbench/plugins/export/__init__.py +0 -0
- reader_workbench/plugins/export/_paths.py +21 -0
- reader_workbench/plugins/export/csv.py +41 -0
- reader_workbench/plugins/export/xlsx.py +44 -0
- reader_workbench/plugins/ingest/__init__.py +0 -0
- reader_workbench/plugins/ingest/_discovery.py +58 -0
- reader_workbench/plugins/ingest/discovery_policy.py +66 -0
- reader_workbench/plugins/ingest/flow_cytometer.py +139 -0
- reader_workbench/plugins/ingest/synergy_h1.py +234 -0
- reader_workbench/plugins/manifests/__init__.py +1 -0
- reader_workbench/plugins/manifests/export.py +29 -0
- reader_workbench/plugins/manifests/ingest.py +29 -0
- reader_workbench/plugins/manifests/plot.py +161 -0
- reader_workbench/plugins/manifests/transform.py +172 -0
- reader_workbench/plugins/manifests/validator.py +18 -0
- reader_workbench/plugins/plot/__init__.py +0 -0
- reader_workbench/plugins/plot/_shared.py +55 -0
- reader_workbench/plugins/plot/cytometry_diagnostic.py +54 -0
- reader_workbench/plugins/plot/distributions.py +58 -0
- reader_workbench/plugins/plot/dual_reporter_triptych.py +224 -0
- reader_workbench/plugins/plot/four_state_event_window_diagnostic.py +133 -0
- reader_workbench/plugins/plot/four_state_event_window_summary.py +53 -0
- reader_workbench/plugins/plot/four_state_vector_collection.py +45 -0
- reader_workbench/plugins/plot/four_state_vector_diagnostic.py +105 -0
- reader_workbench/plugins/plot/four_state_vector_heatmap.py +63 -0
- reader_workbench/plugins/plot/logic_symmetry.py +56 -0
- reader_workbench/plugins/plot/single_reporter_diagnostic.py +279 -0
- reader_workbench/plugins/plot/snapshot_barplot.py +65 -0
- reader_workbench/plugins/plot/snapshot_heatmap.py +104 -0
- reader_workbench/plugins/plot/time_series.py +114 -0
- reader_workbench/plugins/plot/ts_and_snap.py +210 -0
- reader_workbench/plugins/transform/__init__.py +0 -0
- reader_workbench/plugins/transform/_four_state_vector.py +204 -0
- reader_workbench/plugins/transform/_labeling.py +109 -0
- reader_workbench/plugins/transform/alias.py +70 -0
- reader_workbench/plugins/transform/assay_labels.py +62 -0
- reader_workbench/plugins/transform/blank.py +79 -0
- reader_workbench/plugins/transform/crosstalk_pairs.py +180 -0
- reader_workbench/plugins/transform/cytometry_gating.py +120 -0
- reader_workbench/plugins/transform/fold_change.py +79 -0
- reader_workbench/plugins/transform/four_state_event_window.py +93 -0
- reader_workbench/plugins/transform/four_state_vector.py +62 -0
- reader_workbench/plugins/transform/four_state_vector_collection.py +41 -0
- reader_workbench/plugins/transform/logic_symmetry.py +67 -0
- reader_workbench/plugins/transform/outlier_filter.py +60 -0
- reader_workbench/plugins/transform/overflow.py +197 -0
- reader_workbench/plugins/transform/ratio.py +237 -0
- reader_workbench/plugins/transform/sample_map.py +170 -0
- reader_workbench/plugins/transform/sample_metadata.py +94 -0
- reader_workbench/plugins/validator/__init__.py +1 -0
- reader_workbench/plugins/validator/to_tidy_plus_map.py +155 -0
- reader_workbench/protocols/__init__.py +80 -0
- reader_workbench/protocols/_builtins_plate_reader_growth.py +179 -0
- reader_workbench/protocols/_builtins_plate_reader_variants.py +274 -0
- reader_workbench/protocols/builtins.py +1656 -0
- reader_workbench/protocols/compiler.py +22 -0
- reader_workbench/protocols/compilers/__init__.py +1 -0
- reader_workbench/protocols/compilers/common.py +100 -0
- reader_workbench/protocols/compilers/cytometry.py +87 -0
- reader_workbench/protocols/compilers/generic.py +14 -0
- reader_workbench/protocols/compilers/logic.py +245 -0
- reader_workbench/protocols/compilers/plate_reader.py +937 -0
- reader_workbench/protocols/compilers/plate_reader_pipeline.py +197 -0
- reader_workbench/protocols/model.py +1486 -0
- reader_workbench/protocols/semantic_coverage.py +234 -0
- reader_workbench/runtime/__init__.py +12 -0
- reader_workbench/runtime/builtin.py +23 -0
- reader_workbench/runtime/model.py +42 -0
- reader_workbench/workbench/__init__.py +60 -0
- reader_workbench/workbench/assets/__init__.py +22 -0
- reader_workbench/workbench/assets/types.py +118 -0
- reader_workbench/workbench/audit/__init__.py +5 -0
- reader_workbench/workbench/audit/experiments.py +307 -0
- reader_workbench/workbench/audit/staging.py +187 -0
- reader_workbench/workbench/cli/__init__.py +51 -0
- reader_workbench/workbench/cli/_lazy.py +9 -0
- reader_workbench/workbench/cli/_records_view.py +150 -0
- reader_workbench/workbench/cli/_surface_execution.py +443 -0
- reader_workbench/workbench/cli/audit.py +95 -0
- reader_workbench/workbench/cli/automation.py +229 -0
- reader_workbench/workbench/cli/demo.py +46 -0
- reader_workbench/workbench/cli/dop.py +91 -0
- reader_workbench/workbench/cli/experiments.py +635 -0
- reader_workbench/workbench/cli/helpers.py +232 -0
- reader_workbench/workbench/cli/main.py +59 -0
- reader_workbench/workbench/cli/maintenance.py +82 -0
- reader_workbench/workbench/cli/notebooks.py +260 -0
- reader_workbench/workbench/cli/pagination.py +117 -0
- reader_workbench/workbench/cli/protocols.py +336 -0
- reader_workbench/workbench/cli/shared.py +309 -0
- reader_workbench/workbench/cli/surfaces.py +534 -0
- reader_workbench/workbench/cli/verification.py +128 -0
- reader_workbench/workbench/commands.py +10 -0
- reader_workbench/workbench/config/__init__.py +47 -0
- reader_workbench/workbench/config/identity.py +13 -0
- reader_workbench/workbench/config/load.py +405 -0
- reader_workbench/workbench/config/model.py +274 -0
- reader_workbench/workbench/context.py +26 -0
- reader_workbench/workbench/decl/__init__.py +31 -0
- reader_workbench/workbench/decl/build.py +190 -0
- reader_workbench/workbench/decl/model.py +81 -0
- reader_workbench/workbench/dop/__init__.py +12 -0
- reader_workbench/workbench/dop/builtins.py +261 -0
- reader_workbench/workbench/dop/model.py +209 -0
- reader_workbench/workbench/engine/__init__.py +42 -0
- reader_workbench/workbench/engine/_shared.py +76 -0
- reader_workbench/workbench/engine/contracts.py +283 -0
- reader_workbench/workbench/engine/execution.py +326 -0
- reader_workbench/workbench/engine/file_outputs.py +260 -0
- reader_workbench/workbench/engine/inputs.py +161 -0
- reader_workbench/workbench/engine/invocations.py +507 -0
- reader_workbench/workbench/engine/planning.py +72 -0
- reader_workbench/workbench/engine/runtime.py +464 -0
- reader_workbench/workbench/engine/setup.py +149 -0
- reader_workbench/workbench/engine/validation.py +684 -0
- reader_workbench/workbench/experiment/__init__.py +47 -0
- reader_workbench/workbench/experiment/model.py +381 -0
- reader_workbench/workbench/experiments.py +133 -0
- reader_workbench/workbench/graph/__init__.py +47 -0
- reader_workbench/workbench/graph/nodes.py +102 -0
- reader_workbench/workbench/graph/normalize.py +177 -0
- reader_workbench/workbench/graph/refs.py +148 -0
- reader_workbench/workbench/input_discovery.py +19 -0
- reader_workbench/workbench/inspection/__init__.py +3 -0
- reader_workbench/workbench/inspection/catalogs.py +128 -0
- reader_workbench/workbench/inspection/common.py +92 -0
- reader_workbench/workbench/inspection/dop.py +64 -0
- reader_workbench/workbench/inspection/experiments.py +449 -0
- reader_workbench/workbench/inspection/inventory.py +68 -0
- reader_workbench/workbench/inspection/protocols.py +368 -0
- reader_workbench/workbench/inspection/readiness.py +333 -0
- reader_workbench/workbench/inspection/reports.py +367 -0
- reader_workbench/workbench/inspection/results.py +166 -0
- reader_workbench/workbench/inspection/runtime.py +287 -0
- reader_workbench/workbench/inspection/semantics.py +192 -0
- reader_workbench/workbench/inspection/validation.py +30 -0
- reader_workbench/workbench/notebooks/__init__.py +17 -0
- reader_workbench/workbench/notebooks/_launch_registry.py +112 -0
- reader_workbench/workbench/notebooks/_launch_runtime.py +104 -0
- reader_workbench/workbench/notebooks/components/__init__.py +21 -0
- reader_workbench/workbench/notebooks/components/deliverables.py +403 -0
- reader_workbench/workbench/notebooks/components/overview.py +119 -0
- reader_workbench/workbench/notebooks/eda.marimo.py.txt +153 -0
- reader_workbench/workbench/notebooks/launch.py +274 -0
- reader_workbench/workbench/notebooks/presentation.py +136 -0
- reader_workbench/workbench/notebooks/scaffold.py +60 -0
- reader_workbench/workbench/ontology.py +78 -0
- reader_workbench/workbench/paths.py +44 -0
- reader_workbench/workbench/ports/__init__.py +31 -0
- reader_workbench/workbench/ports/model.py +168 -0
- reader_workbench/workbench/records/__init__.py +44 -0
- reader_workbench/workbench/records/epoch.py +329 -0
- reader_workbench/workbench/records/evidence.py +247 -0
- reader_workbench/workbench/records/identity.py +87 -0
- reader_workbench/workbench/records/locking.py +185 -0
- reader_workbench/workbench/records/model.py +711 -0
- reader_workbench/workbench/records/sources.py +73 -0
- reader_workbench/workbench/records/store.py +1022 -0
- reader_workbench/workbench/records/verification.py +998 -0
- reader_workbench/workbench/registry.py +333 -0
- reader_workbench/workbench/spec_overrides.py +215 -0
- reader_workbench-1.0.0.dist-info/METADATA +91 -0
- reader_workbench-1.0.0.dist-info/RECORD +293 -0
- reader_workbench-1.0.0.dist-info/WHEEL +5 -0
- reader_workbench-1.0.0.dist-info/entry_points.txt +2 -0
- reader_workbench-1.0.0.dist-info/licenses/LICENSE +21 -0
- reader_workbench-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
"""Explicit dataframe-contract catalog and contract-surface helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterable, Iterator, Mapping
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
from reader_workbench.errors import ContractError
|
|
11
|
+
|
|
12
|
+
from .model import ColumnRule, ContractId, ContractToken, DataFrameContract, DType, validate_df
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _dtype_satisfies(*, child: DType, parent: DType) -> bool:
|
|
16
|
+
"""Return whether every value accepted by a child dtype is accepted by its parent."""
|
|
17
|
+
|
|
18
|
+
return child == parent or (child == "int" and parent == "float")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _require_compatible_rule(
|
|
22
|
+
*,
|
|
23
|
+
child_contract: DataFrameContract,
|
|
24
|
+
parent_contract: DataFrameContract,
|
|
25
|
+
child_rule: ColumnRule,
|
|
26
|
+
parent_rule: ColumnRule,
|
|
27
|
+
) -> None:
|
|
28
|
+
prefix = (
|
|
29
|
+
f"contract {child_contract.id!r} is not structurally compatible with parent "
|
|
30
|
+
f"{parent_contract.id!r}: column {parent_rule.name!r}"
|
|
31
|
+
)
|
|
32
|
+
if not _dtype_satisfies(child=child_rule.dtype, parent=parent_rule.dtype):
|
|
33
|
+
raise ContractError(f"{prefix} has dtype {child_rule.dtype!r}, expected a subtype of {parent_rule.dtype!r}")
|
|
34
|
+
if not parent_rule.allow_nan and child_rule.allow_nan:
|
|
35
|
+
raise ContractError(f"{prefix} permits null values forbidden by the parent")
|
|
36
|
+
if parent_rule.monotone_non_decreasing and not child_rule.monotone_non_decreasing:
|
|
37
|
+
raise ContractError(f"{prefix} does not preserve the parent's monotone constraint")
|
|
38
|
+
if parent_rule.nonnegative and not child_rule.nonnegative:
|
|
39
|
+
raise ContractError(f"{prefix} does not preserve the parent's nonnegative constraint")
|
|
40
|
+
if parent_rule.allowed_values is not None:
|
|
41
|
+
if child_rule.allowed_values is None:
|
|
42
|
+
raise ContractError(f"{prefix} does not preserve the parent's allowed-values constraint")
|
|
43
|
+
parent_values = {str(value) for value in parent_rule.allowed_values}
|
|
44
|
+
child_values = {str(value) for value in child_rule.allowed_values}
|
|
45
|
+
if not child_values <= parent_values:
|
|
46
|
+
raise ContractError(f"{prefix} permits values outside the parent's allowed set")
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _require_structural_compatibility(*, child: DataFrameContract, parent: DataFrameContract) -> None:
|
|
50
|
+
child_columns = {rule.name: rule for rule in child.columns}
|
|
51
|
+
parent_columns = {rule.name: rule for rule in parent.columns}
|
|
52
|
+
|
|
53
|
+
for name, parent_rule in parent_columns.items():
|
|
54
|
+
child_rule = child_columns.get(name)
|
|
55
|
+
if child_rule is None:
|
|
56
|
+
if parent_rule.required:
|
|
57
|
+
raise ContractError(
|
|
58
|
+
f"contract {child.id!r} is not structurally compatible with parent {parent.id!r}: "
|
|
59
|
+
f"missing required column {name!r}"
|
|
60
|
+
)
|
|
61
|
+
if child.allow_extra_columns:
|
|
62
|
+
raise ContractError(
|
|
63
|
+
f"contract {child.id!r} is not structurally compatible with parent {parent.id!r}: "
|
|
64
|
+
f"optional parent column {name!r} is unconstrained by the child"
|
|
65
|
+
)
|
|
66
|
+
continue
|
|
67
|
+
if parent_rule.required and not child_rule.required:
|
|
68
|
+
raise ContractError(
|
|
69
|
+
f"contract {child.id!r} is not structurally compatible with parent {parent.id!r}: "
|
|
70
|
+
f"required column {name!r} is optional in the child"
|
|
71
|
+
)
|
|
72
|
+
_require_compatible_rule(
|
|
73
|
+
child_contract=child,
|
|
74
|
+
parent_contract=parent,
|
|
75
|
+
child_rule=child_rule,
|
|
76
|
+
parent_rule=parent_rule,
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
if not parent.allow_extra_columns:
|
|
80
|
+
extras = sorted(set(child_columns) - set(parent_columns))
|
|
81
|
+
if child.allow_extra_columns or extras:
|
|
82
|
+
detail = "allows undeclared columns" if child.allow_extra_columns else f"declares extra columns {extras}"
|
|
83
|
+
raise ContractError(
|
|
84
|
+
f"contract {child.id!r} is not structurally compatible with parent {parent.id!r}: {detail}"
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
child_keys = [set(key) for key in child.unique_keys if key]
|
|
88
|
+
for parent_key in (set(key) for key in parent.unique_keys if key):
|
|
89
|
+
if not any(child_key <= parent_key for child_key in child_keys):
|
|
90
|
+
raise ContractError(
|
|
91
|
+
f"contract {child.id!r} is not structurally compatible with parent {parent.id!r}: "
|
|
92
|
+
f"does not preserve unique key {sorted(parent_key)}"
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _validate_contract_graph(contracts: Mapping[ContractId, DataFrameContract]) -> None:
|
|
97
|
+
for contract_id, contract in contracts.items():
|
|
98
|
+
if not isinstance(contract, DataFrameContract):
|
|
99
|
+
raise ContractError(f"contract catalog entry {contract_id!r} must be a DataFrameContract")
|
|
100
|
+
if not isinstance(contract.id, str) or not contract.id.strip():
|
|
101
|
+
raise ContractError("contract ids must be non-empty strings")
|
|
102
|
+
if contract.id != contract_id:
|
|
103
|
+
raise ContractError(f"contract catalog key {contract_id!r} does not match contract id {contract.id!r}")
|
|
104
|
+
if any(not isinstance(rule, ColumnRule) for rule in contract.columns):
|
|
105
|
+
raise ContractError(f"contract {contract.id!r} columns must contain ColumnRule values")
|
|
106
|
+
column_names = [rule.name for rule in contract.columns]
|
|
107
|
+
if any(not isinstance(name, str) or not name.strip() for name in column_names):
|
|
108
|
+
raise ContractError(f"contract {contract.id!r} column names must be non-empty strings")
|
|
109
|
+
if len(column_names) != len(set(column_names)):
|
|
110
|
+
raise ContractError(f"contract {contract.id!r} declares duplicate column names")
|
|
111
|
+
if type(contract.allow_extra_columns) is not bool:
|
|
112
|
+
raise ContractError(f"contract {contract.id!r} allow_extra_columns must be bool")
|
|
113
|
+
columns_by_name = {rule.name: rule for rule in contract.columns}
|
|
114
|
+
for rule in contract.columns:
|
|
115
|
+
for field in ("required", "allow_nan", "monotone_non_decreasing", "nonnegative"):
|
|
116
|
+
if type(getattr(rule, field)) is not bool:
|
|
117
|
+
raise ContractError(f"contract {contract.id!r} column {rule.name!r} {field} must be bool")
|
|
118
|
+
if rule.nonnegative and rule.dtype not in {"int", "float"}:
|
|
119
|
+
raise ContractError(
|
|
120
|
+
f"contract {contract.id!r} column {rule.name!r} uses nonnegative with non-numeric dtype"
|
|
121
|
+
)
|
|
122
|
+
if rule.monotone_non_decreasing and rule.dtype not in {"int", "float", "datetime"}:
|
|
123
|
+
raise ContractError(
|
|
124
|
+
f"contract {contract.id!r} column {rule.name!r} uses monotonicity with unordered dtype"
|
|
125
|
+
)
|
|
126
|
+
for key in contract.unique_keys:
|
|
127
|
+
if not key:
|
|
128
|
+
raise ContractError(f"contract {contract.id!r} unique keys must not be empty")
|
|
129
|
+
if len(key) != len(set(key)):
|
|
130
|
+
raise ContractError(f"contract {contract.id!r} unique key {list(key)!r} contains duplicate columns")
|
|
131
|
+
for name in key:
|
|
132
|
+
rule = columns_by_name.get(name)
|
|
133
|
+
if rule is None:
|
|
134
|
+
raise ContractError(f"contract {contract.id!r} unique key references unknown column {name!r}")
|
|
135
|
+
if not rule.required:
|
|
136
|
+
raise ContractError(f"contract {contract.id!r} unique key column {name!r} must be required")
|
|
137
|
+
if contract.primary_index is not None:
|
|
138
|
+
if not contract.primary_index:
|
|
139
|
+
raise ContractError(f"contract {contract.id!r} primary index must not be empty")
|
|
140
|
+
if len(contract.primary_index) != len(set(contract.primary_index)):
|
|
141
|
+
raise ContractError(f"contract {contract.id!r} primary index contains duplicate columns")
|
|
142
|
+
for name in contract.primary_index:
|
|
143
|
+
rule = columns_by_name.get(name)
|
|
144
|
+
if rule is None:
|
|
145
|
+
raise ContractError(f"contract {contract.id!r} primary index references unknown column {name!r}")
|
|
146
|
+
if not rule.required:
|
|
147
|
+
raise ContractError(f"contract {contract.id!r} primary index column {name!r} must be required")
|
|
148
|
+
for parent in contract.parents:
|
|
149
|
+
if parent == contract.id:
|
|
150
|
+
raise ContractError(f"contract {contract.id!r} cannot list itself as a parent")
|
|
151
|
+
if parent not in contracts:
|
|
152
|
+
raise ContractError(f"contract {contract.id!r} references unknown parent {parent!r}")
|
|
153
|
+
|
|
154
|
+
visited: set[ContractId] = set()
|
|
155
|
+
active: set[ContractId] = set()
|
|
156
|
+
path: list[ContractId] = []
|
|
157
|
+
|
|
158
|
+
def visit(contract_id: ContractId) -> None:
|
|
159
|
+
if contract_id in active:
|
|
160
|
+
cycle_start = path.index(contract_id)
|
|
161
|
+
cycle = [*path[cycle_start:], contract_id]
|
|
162
|
+
raise ContractError("contract lineage cycle detected: " + " -> ".join(cycle))
|
|
163
|
+
if contract_id in visited:
|
|
164
|
+
return
|
|
165
|
+
active.add(contract_id)
|
|
166
|
+
path.append(contract_id)
|
|
167
|
+
for parent in contracts[contract_id].parents:
|
|
168
|
+
visit(parent)
|
|
169
|
+
path.pop()
|
|
170
|
+
active.remove(contract_id)
|
|
171
|
+
visited.add(contract_id)
|
|
172
|
+
|
|
173
|
+
for contract_id in sorted(contracts):
|
|
174
|
+
visit(contract_id)
|
|
175
|
+
|
|
176
|
+
for child in contracts.values():
|
|
177
|
+
for parent_id in child.parents:
|
|
178
|
+
_require_structural_compatibility(child=child, parent=contracts[parent_id])
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
@dataclass(frozen=True)
|
|
182
|
+
class OutputContractSurface:
|
|
183
|
+
minimum: ContractToken
|
|
184
|
+
runtime_mode: str = "fixed" # fixed | promoted | passthrough
|
|
185
|
+
promoted: tuple[ContractId, ...] = ()
|
|
186
|
+
note: str | None = None
|
|
187
|
+
|
|
188
|
+
def render(self) -> str:
|
|
189
|
+
if self.minimum == "none":
|
|
190
|
+
return "none"
|
|
191
|
+
parts = [self.minimum]
|
|
192
|
+
if self.runtime_mode == "promoted" and self.promoted:
|
|
193
|
+
parts.append(f"runtime may promote to {', '.join(self.promoted)}")
|
|
194
|
+
elif self.runtime_mode == "passthrough":
|
|
195
|
+
hint = f" (e.g. {', '.join(self.promoted)})" if self.promoted else ""
|
|
196
|
+
parts.append(f"runtime preserves stricter compatible input contracts{hint}")
|
|
197
|
+
if self.note:
|
|
198
|
+
parts.append(self.note)
|
|
199
|
+
return "; ".join(parts)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
class ContractCatalog:
|
|
203
|
+
"""Immutable contract catalog with explicit lineage and validation helpers."""
|
|
204
|
+
|
|
205
|
+
def __init__(self, contracts: Mapping[ContractId, DataFrameContract]) -> None:
|
|
206
|
+
normalized = dict(contracts)
|
|
207
|
+
_validate_contract_graph(normalized)
|
|
208
|
+
self._contracts = normalized
|
|
209
|
+
|
|
210
|
+
@classmethod
|
|
211
|
+
def from_contracts(cls, contracts: Iterable[DataFrameContract]) -> ContractCatalog:
|
|
212
|
+
normalized: dict[ContractId, DataFrameContract] = {}
|
|
213
|
+
for contract in contracts:
|
|
214
|
+
if contract.id in normalized:
|
|
215
|
+
raise ContractError(f"duplicate contract id {contract.id!r}")
|
|
216
|
+
normalized[contract.id] = contract
|
|
217
|
+
return cls(normalized)
|
|
218
|
+
|
|
219
|
+
def all(self) -> tuple[DataFrameContract, ...]:
|
|
220
|
+
return tuple(self._contracts[key] for key in sorted(self._contracts))
|
|
221
|
+
|
|
222
|
+
def ids(self) -> tuple[ContractId, ...]:
|
|
223
|
+
return tuple(sorted(self._contracts))
|
|
224
|
+
|
|
225
|
+
def __contains__(self, contract_id: object) -> bool:
|
|
226
|
+
return isinstance(contract_id, str) and contract_id in self._contracts
|
|
227
|
+
|
|
228
|
+
def get(self, contract_id: ContractToken | None) -> DataFrameContract | None:
|
|
229
|
+
if contract_id in (None, "none"):
|
|
230
|
+
return None
|
|
231
|
+
return self._contracts.get(contract_id)
|
|
232
|
+
|
|
233
|
+
def require(self, contract_id: ContractId) -> DataFrameContract:
|
|
234
|
+
try:
|
|
235
|
+
return self._contracts[contract_id]
|
|
236
|
+
except KeyError as exc:
|
|
237
|
+
raise ContractError(f"unknown contract id {contract_id!r}") from exc
|
|
238
|
+
|
|
239
|
+
def iter_lineage(self, contract_id: ContractId) -> Iterator[ContractId]:
|
|
240
|
+
seen: set[ContractId] = set()
|
|
241
|
+
stack: list[ContractId] = [contract_id]
|
|
242
|
+
while stack:
|
|
243
|
+
current = stack.pop()
|
|
244
|
+
if current in seen:
|
|
245
|
+
continue
|
|
246
|
+
contract = self.require(current)
|
|
247
|
+
seen.add(current)
|
|
248
|
+
yield current
|
|
249
|
+
stack.extend(reversed(contract.parents))
|
|
250
|
+
|
|
251
|
+
def satisfies(self, *, actual: ContractToken | None, expected: ContractId) -> bool:
|
|
252
|
+
if actual in (None, "none"):
|
|
253
|
+
return False
|
|
254
|
+
return expected in set(self.iter_lineage(actual))
|
|
255
|
+
|
|
256
|
+
def validate(self, df: pd.DataFrame, *, contract_id: ContractId, where: str) -> None:
|
|
257
|
+
validate_df(df, self.require(contract_id), where=where)
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Literal
|
|
5
|
+
|
|
6
|
+
import pandas as pd
|
|
7
|
+
|
|
8
|
+
from reader_workbench.errors import ContractError
|
|
9
|
+
|
|
10
|
+
type ContractId = str
|
|
11
|
+
DType = Literal["string", "int", "float", "bool", "category", "datetime"]
|
|
12
|
+
type ContractToken = ContractId | Literal["none"]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class ColumnRule:
|
|
17
|
+
name: str
|
|
18
|
+
dtype: DType
|
|
19
|
+
required: bool = True
|
|
20
|
+
allow_nan: bool = False
|
|
21
|
+
monotone_non_decreasing: bool = False
|
|
22
|
+
nonnegative: bool = False
|
|
23
|
+
allowed_values: tuple[str, ...] | None = None
|
|
24
|
+
|
|
25
|
+
def __post_init__(self) -> None:
|
|
26
|
+
if self.allowed_values is not None:
|
|
27
|
+
object.__setattr__(self, "allowed_values", tuple(self.allowed_values))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class DataFrameContract:
|
|
32
|
+
id: ContractId
|
|
33
|
+
description: str
|
|
34
|
+
columns: tuple[ColumnRule, ...]
|
|
35
|
+
unique_keys: tuple[tuple[str, ...], ...]
|
|
36
|
+
parents: tuple[ContractId, ...] = ()
|
|
37
|
+
domain: str | None = None
|
|
38
|
+
kind: str | None = None
|
|
39
|
+
primary_index: tuple[str, ...] | None = None
|
|
40
|
+
notes: str | None = None
|
|
41
|
+
allow_extra_columns: bool = True
|
|
42
|
+
|
|
43
|
+
def __post_init__(self) -> None:
|
|
44
|
+
object.__setattr__(self, "columns", tuple(self.columns))
|
|
45
|
+
object.__setattr__(self, "unique_keys", tuple(tuple(key) for key in self.unique_keys))
|
|
46
|
+
object.__setattr__(self, "parents", tuple(self.parents))
|
|
47
|
+
if self.primary_index is not None:
|
|
48
|
+
object.__setattr__(self, "primary_index", tuple(self.primary_index))
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _is_dtype(series: pd.Series, want: DType) -> bool:
|
|
52
|
+
if want == "string":
|
|
53
|
+
return pd.api.types.is_string_dtype(series) or pd.api.types.is_object_dtype(series)
|
|
54
|
+
if want == "int":
|
|
55
|
+
return pd.api.types.is_integer_dtype(series)
|
|
56
|
+
if want == "float":
|
|
57
|
+
return pd.api.types.is_float_dtype(series) or pd.api.types.is_integer_dtype(series)
|
|
58
|
+
if want == "bool":
|
|
59
|
+
return pd.api.types.is_bool_dtype(series)
|
|
60
|
+
if want == "category":
|
|
61
|
+
return pd.api.types.is_categorical_dtype(series)
|
|
62
|
+
if want == "datetime":
|
|
63
|
+
return pd.api.types.is_datetime64_any_dtype(series)
|
|
64
|
+
return False
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def validate_df(df: pd.DataFrame, contract: DataFrameContract, *, where: str) -> None:
|
|
68
|
+
"""Assert df matches the contract exactly; raise ContractError on first failure."""
|
|
69
|
+
cols = set(df.columns)
|
|
70
|
+
declared = {rule.name for rule in contract.columns}
|
|
71
|
+
if not contract.allow_extra_columns:
|
|
72
|
+
extras = sorted(cols - declared)
|
|
73
|
+
if extras:
|
|
74
|
+
raise ContractError(f"[{where}] contract {contract.id}: unexpected columns {extras}")
|
|
75
|
+
|
|
76
|
+
for rule in contract.columns:
|
|
77
|
+
if rule.required and rule.name not in cols:
|
|
78
|
+
raise ContractError(f"[{where}] contract {contract.id}: missing required column '{rule.name}'")
|
|
79
|
+
|
|
80
|
+
for rule in contract.columns:
|
|
81
|
+
if rule.name not in cols:
|
|
82
|
+
continue
|
|
83
|
+
s = df[rule.name]
|
|
84
|
+
if not _is_dtype(s, rule.dtype):
|
|
85
|
+
raise ContractError(
|
|
86
|
+
f"[{where}] contract {contract.id}: column '{rule.name}' has dtype {s.dtype} but expected {rule.dtype}"
|
|
87
|
+
)
|
|
88
|
+
if not rule.allow_nan and s.isna().any():
|
|
89
|
+
raise ContractError(
|
|
90
|
+
f"[{where}] contract {contract.id}: column '{rule.name}' contains NaN but allow_nan=false"
|
|
91
|
+
)
|
|
92
|
+
if rule.nonnegative and (pd.to_numeric(s, errors="coerce") < 0).any():
|
|
93
|
+
raise ContractError(f"[{where}] contract {contract.id}: column '{rule.name}' must be nonnegative")
|
|
94
|
+
if rule.monotone_non_decreasing and not s.dropna().is_monotonic_increasing:
|
|
95
|
+
raise ContractError(
|
|
96
|
+
f"[{where}] contract {contract.id}: column '{rule.name}' must be monotone non-decreasing"
|
|
97
|
+
)
|
|
98
|
+
if rule.allowed_values is not None:
|
|
99
|
+
bad = sorted(set(map(str, s.dropna().astype(str))) - set(map(str, rule.allowed_values)))
|
|
100
|
+
if bad:
|
|
101
|
+
raise ContractError(
|
|
102
|
+
f"[{where}] contract {contract.id}: column '{rule.name}' contains values outside allowed set: {bad[:5]}"
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
for key in contract.unique_keys:
|
|
106
|
+
if not key:
|
|
107
|
+
continue
|
|
108
|
+
if df.duplicated(subset=key, keep=False).any():
|
|
109
|
+
raise ContractError(f"[{where}] contract {contract.id}: uniqueness violated for key {key}")
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Domain-owned protocol and data semantics."""
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Public event-table preparation and analysis for cytometry workflows."""
|
|
2
|
+
|
|
3
|
+
from reader_workbench.domains.cytometry.analysis.events import (
|
|
4
|
+
CytometryAnalysis,
|
|
5
|
+
CytometryAnalysisError,
|
|
6
|
+
GateSpec,
|
|
7
|
+
ThresholdSpec,
|
|
8
|
+
prepare_event_table,
|
|
9
|
+
)
|
|
10
|
+
from reader_workbench.domains.cytometry.analysis.gating import analyze_events
|
|
11
|
+
from reader_workbench.domains.cytometry.analysis.workflow import (
|
|
12
|
+
CytometryGatingRequest,
|
|
13
|
+
CytometryGatingResult,
|
|
14
|
+
CytometryQCSpec,
|
|
15
|
+
run_cytometry_gating,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"CytometryAnalysis",
|
|
20
|
+
"CytometryAnalysisError",
|
|
21
|
+
"CytometryGatingRequest",
|
|
22
|
+
"CytometryGatingResult",
|
|
23
|
+
"CytometryQCSpec",
|
|
24
|
+
"GateSpec",
|
|
25
|
+
"ThresholdSpec",
|
|
26
|
+
"analyze_events",
|
|
27
|
+
"prepare_event_table",
|
|
28
|
+
"run_cytometry_gating",
|
|
29
|
+
]
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
"""Polars-native preparation for tidy cytometry event tables."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Sequence
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Literal
|
|
8
|
+
|
|
9
|
+
import polars as pl
|
|
10
|
+
|
|
11
|
+
_REQUIRED_TIDY_COLUMNS = ("channel", "value", "sample_id", "event_index")
|
|
12
|
+
_METADATA_COLUMNS = ("treatment", "design_id", "sample_label")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class CytometryAnalysisError(ValueError):
|
|
16
|
+
"""Raised when a cytometry event table cannot support the requested analysis."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True, slots=True)
|
|
20
|
+
class GateSpec:
|
|
21
|
+
"""Sequential cells and singlets gates for an event-wide table."""
|
|
22
|
+
|
|
23
|
+
cells_x_channel: str
|
|
24
|
+
cells_y_channel: str
|
|
25
|
+
cells_x_range: tuple[float, float]
|
|
26
|
+
cells_y_range: tuple[float, float]
|
|
27
|
+
singlet_x_channel: str
|
|
28
|
+
singlet_y_channel: str
|
|
29
|
+
singlet_ratio_range: tuple[float, float]
|
|
30
|
+
cells_enabled: bool = True
|
|
31
|
+
singlets_enabled: bool = True
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass(frozen=True, slots=True)
|
|
35
|
+
class ThresholdSpec:
|
|
36
|
+
"""Positive-event threshold applied to the selected fluorescence channel."""
|
|
37
|
+
|
|
38
|
+
channel: str
|
|
39
|
+
value: float = 0.0
|
|
40
|
+
mode: Literal["manual", "from_control_quantile"] = "manual"
|
|
41
|
+
group_column: str | None = None
|
|
42
|
+
control_value: str | None = None
|
|
43
|
+
quantile: float = 0.99
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True, slots=True)
|
|
47
|
+
class CytometryAnalysis:
|
|
48
|
+
"""Polars-native event, count, summary, and quality-control tables."""
|
|
49
|
+
|
|
50
|
+
gated_events: pl.DataFrame
|
|
51
|
+
gate_counts_sample: pl.DataFrame
|
|
52
|
+
stats_sample: pl.DataFrame
|
|
53
|
+
stats_group: pl.DataFrame | None
|
|
54
|
+
qc_table: pl.DataFrame
|
|
55
|
+
threshold_value: float
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
EventFrame = pl.DataFrame | pl.LazyFrame
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def prepare_event_table(
|
|
62
|
+
frame: EventFrame,
|
|
63
|
+
*,
|
|
64
|
+
channels: Sequence[str],
|
|
65
|
+
metadata_columns: Sequence[str] = (),
|
|
66
|
+
) -> pl.DataFrame:
|
|
67
|
+
"""Pivot declared channels from a tidy event table without inferring policy."""
|
|
68
|
+
|
|
69
|
+
available = _frame_columns(frame)
|
|
70
|
+
missing_tidy = [column for column in _REQUIRED_TIDY_COLUMNS if column not in available]
|
|
71
|
+
if missing_tidy:
|
|
72
|
+
raise CytometryAnalysisError(
|
|
73
|
+
"Cytometry event data is missing required column(s): " + ", ".join(missing_tidy) + "."
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
selected_channels = tuple(dict.fromkeys(str(channel) for channel in channels if str(channel)))
|
|
77
|
+
if not selected_channels:
|
|
78
|
+
raise CytometryAnalysisError("Select at least one cytometry channel.")
|
|
79
|
+
|
|
80
|
+
requested_metadata = tuple(dict.fromkeys(str(column).strip() for column in metadata_columns if str(column).strip()))
|
|
81
|
+
missing_metadata = [column for column in requested_metadata if column not in available]
|
|
82
|
+
if missing_metadata:
|
|
83
|
+
raise CytometryAnalysisError(
|
|
84
|
+
"Missing requested cytometry metadata column(s): " + ", ".join(missing_metadata) + "."
|
|
85
|
+
)
|
|
86
|
+
retained_metadata = list(
|
|
87
|
+
dict.fromkeys((*[column for column in _METADATA_COLUMNS if column in available], *requested_metadata))
|
|
88
|
+
)
|
|
89
|
+
index_columns = ["sample_id", "event_index", *retained_metadata]
|
|
90
|
+
pivot_key_columns = [*index_columns, "channel"]
|
|
91
|
+
projected_columns = [*index_columns, "channel", "value"]
|
|
92
|
+
event_identity_columns = ["sample_id", "event_index"]
|
|
93
|
+
metadata_consistency_columns = {
|
|
94
|
+
column: f"__reader_event_metadata_n_unique__{column}" for column in retained_metadata
|
|
95
|
+
}
|
|
96
|
+
query = _as_lazy(frame).select(projected_columns)
|
|
97
|
+
if metadata_consistency_columns:
|
|
98
|
+
query = query.with_columns(
|
|
99
|
+
*[
|
|
100
|
+
pl.col(column).n_unique().over(event_identity_columns).alias(consistency_column)
|
|
101
|
+
for column, consistency_column in metadata_consistency_columns.items()
|
|
102
|
+
]
|
|
103
|
+
)
|
|
104
|
+
query = query.filter(pl.col("channel").is_in(selected_channels))
|
|
105
|
+
|
|
106
|
+
channel_profile = query.select(
|
|
107
|
+
pl.len().alias("row_count"),
|
|
108
|
+
pl.struct(pivot_key_columns).n_unique().alias("unique_pivot_key_count"),
|
|
109
|
+
pl.col("channel").unique().sort().implode().alias("channels"),
|
|
110
|
+
pl.col("channel").n_unique().over(event_identity_columns).min().alias("minimum_event_selected_channel_count"),
|
|
111
|
+
*[pl.col(consistency_column).max() for consistency_column in metadata_consistency_columns.values()],
|
|
112
|
+
).collect()
|
|
113
|
+
row_count = int(channel_profile.item(0, "row_count"))
|
|
114
|
+
unique_pivot_key_count = int(channel_profile.item(0, "unique_pivot_key_count"))
|
|
115
|
+
if row_count:
|
|
116
|
+
inconsistent_metadata = [
|
|
117
|
+
column
|
|
118
|
+
for column, consistency_column in metadata_consistency_columns.items()
|
|
119
|
+
if int(channel_profile.item(0, consistency_column)) > 1
|
|
120
|
+
]
|
|
121
|
+
if inconsistent_metadata:
|
|
122
|
+
identity = ", ".join(event_identity_columns)
|
|
123
|
+
metadata = ", ".join(inconsistent_metadata)
|
|
124
|
+
raise CytometryAnalysisError(
|
|
125
|
+
"Cytometry event data contains inconsistent metadata across channel rows for "
|
|
126
|
+
f"event identity {identity}: {metadata}."
|
|
127
|
+
)
|
|
128
|
+
if row_count > unique_pivot_key_count:
|
|
129
|
+
duplicate_row_count = row_count - unique_pivot_key_count
|
|
130
|
+
keys = ", ".join(pivot_key_columns)
|
|
131
|
+
raise CytometryAnalysisError(
|
|
132
|
+
f"Cytometry event data contains {duplicate_row_count} duplicate pivot key rows across "
|
|
133
|
+
f"{keys}; each event/channel key must be unique."
|
|
134
|
+
)
|
|
135
|
+
if row_count:
|
|
136
|
+
present_channels = set(channel_profile.item(0, "channels"))
|
|
137
|
+
missing_channels = [channel for channel in selected_channels if channel not in present_channels]
|
|
138
|
+
if missing_channels:
|
|
139
|
+
raise CytometryAnalysisError("Missing channels after pivot: " + ", ".join(missing_channels) + ".")
|
|
140
|
+
minimum_channel_count = int(channel_profile.item(0, "minimum_event_selected_channel_count"))
|
|
141
|
+
if minimum_channel_count < len(selected_channels):
|
|
142
|
+
identity = ", ".join(event_identity_columns)
|
|
143
|
+
required_channels = ", ".join(selected_channels)
|
|
144
|
+
raise CytometryAnalysisError(
|
|
145
|
+
"Cytometry event data is missing selected channels within at least one event; "
|
|
146
|
+
f"each {identity} must contain all of: {required_channels}."
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
return (
|
|
150
|
+
query.select(projected_columns)
|
|
151
|
+
.with_columns(pl.col("value").cast(pl.Float64))
|
|
152
|
+
.pivot(
|
|
153
|
+
on="channel",
|
|
154
|
+
on_columns=selected_channels,
|
|
155
|
+
values="value",
|
|
156
|
+
index=index_columns,
|
|
157
|
+
maintain_order=True,
|
|
158
|
+
)
|
|
159
|
+
.collect()
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _as_lazy(frame: EventFrame) -> pl.LazyFrame:
|
|
164
|
+
if isinstance(frame, pl.LazyFrame):
|
|
165
|
+
return frame
|
|
166
|
+
if isinstance(frame, pl.DataFrame):
|
|
167
|
+
return frame.lazy()
|
|
168
|
+
raise TypeError(f"Expected a Polars DataFrame or LazyFrame, got {type(frame).__name__}.")
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _frame_columns(frame: EventFrame) -> tuple[str, ...]:
|
|
172
|
+
if isinstance(frame, pl.LazyFrame):
|
|
173
|
+
return tuple(frame.collect_schema().names())
|
|
174
|
+
if isinstance(frame, pl.DataFrame):
|
|
175
|
+
return tuple(frame.columns)
|
|
176
|
+
raise TypeError(f"Expected a Polars DataFrame or LazyFrame, got {type(frame).__name__}.")
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _require_columns(frame: pl.DataFrame, columns: Sequence[str]) -> None:
|
|
180
|
+
missing = [column for column in dict.fromkeys(columns) if column not in frame.columns]
|
|
181
|
+
if missing:
|
|
182
|
+
raise CytometryAnalysisError("Missing cytometry column(s): " + ", ".join(missing) + ".")
|