reader-workbench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- reader_workbench/__init__.py +22 -0
- reader_workbench/__main__.py +4 -0
- reader_workbench/_version.py +17 -0
- reader_workbench/api/__init__.py +74 -0
- reader_workbench/api/_record_reads.py +75 -0
- reader_workbench/api/artifacts.py +79 -0
- reader_workbench/api/facade.py +538 -0
- reader_workbench/api/models.py +285 -0
- reader_workbench/api/notebooks.py +63 -0
- reader_workbench/contracts/__init__.py +18 -0
- reader_workbench/contracts/builtins/__init__.py +36 -0
- reader_workbench/contracts/builtins/cytometry.py +140 -0
- reader_workbench/contracts/builtins/four_state_event_window.py +214 -0
- reader_workbench/contracts/builtins/generic.py +21 -0
- reader_workbench/contracts/builtins/logic.py +149 -0
- reader_workbench/contracts/builtins/plate_reader.py +47 -0
- reader_workbench/contracts/catalog.py +257 -0
- reader_workbench/contracts/model.py +109 -0
- reader_workbench/domains/__init__.py +1 -0
- reader_workbench/domains/cytometry/__init__.py +3 -0
- reader_workbench/domains/cytometry/analysis/__init__.py +29 -0
- reader_workbench/domains/cytometry/analysis/events.py +182 -0
- reader_workbench/domains/cytometry/analysis/gating.py +175 -0
- reader_workbench/domains/cytometry/analysis/workflow.py +248 -0
- reader_workbench/domains/cytometry/io/__init__.py +3 -0
- reader_workbench/domains/cytometry/io/fcs.py +135 -0
- reader_workbench/domains/cytometry/plots/__init__.py +5 -0
- reader_workbench/domains/cytometry/plots/diagnostic.py +155 -0
- reader_workbench/domains/logic/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/pairs.py +661 -0
- reader_workbench/domains/logic/four_state_vector/__init__.py +7 -0
- reader_workbench/domains/logic/four_state_vector/builder.py +321 -0
- reader_workbench/domains/logic/four_state_vector/collection/__init__.py +20 -0
- reader_workbench/domains/logic/four_state_vector/collection/checks.py +49 -0
- reader_workbench/domains/logic/four_state_vector/collection/constants.py +29 -0
- reader_workbench/domains/logic/four_state_vector/collection/model.py +19 -0
- reader_workbench/domains/logic/four_state_vector/collection/render.py +383 -0
- reader_workbench/domains/logic/four_state_vector/collection/sources.py +185 -0
- reader_workbench/domains/logic/four_state_vector/config.py +214 -0
- reader_workbench/domains/logic/four_state_vector/diagnostic.py +361 -0
- reader_workbench/domains/logic/four_state_vector/heatmap.py +86 -0
- reader_workbench/domains/logic/four_state_vector/math.py +191 -0
- reader_workbench/domains/logic/four_state_vector/reference.py +85 -0
- reader_workbench/domains/logic/four_state_vector/selection.py +228 -0
- reader_workbench/domains/logic/four_state_vector/treatment_semantics.py +51 -0
- reader_workbench/domains/logic/four_state_vector/validation.py +19 -0
- reader_workbench/domains/logic/logic_symmetry/__init__.py +3 -0
- reader_workbench/domains/logic/logic_symmetry/encodings.py +93 -0
- reader_workbench/domains/logic/logic_symmetry/extract_corners.py +192 -0
- reader_workbench/domains/logic/logic_symmetry/main.py +236 -0
- reader_workbench/domains/logic/logic_symmetry/metrics.py +96 -0
- reader_workbench/domains/logic/logic_symmetry/overlay.py +129 -0
- reader_workbench/domains/logic/logic_symmetry/prep.py +138 -0
- reader_workbench/domains/logic/logic_symmetry/render.py +356 -0
- reader_workbench/domains/logic/treatment_columns.py +42 -0
- reader_workbench/domains/plate_reader/__init__.py +1 -0
- reader_workbench/domains/plate_reader/analysis/__init__.py +14 -0
- reader_workbench/domains/plate_reader/analysis/fold_change.py +474 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/__init__.py +21 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/aggregation.py +191 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contract_fields.py +50 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contracts.py +320 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/design_dispositions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/disposition_records.py +140 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/event_sensitivity.py +27 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/materialize.py +274 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/observation_resampling.py +97 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/reduction.py +62 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/seeds.py +15 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/sources.py +308 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusion_validation.py +94 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/timepoints.py +76 -0
- reader_workbench/domains/plate_reader/io/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/sample_map.py +65 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_kinetic.py +141 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_parser.py +295 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_shared.py +195 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_snapshot.py +130 -0
- reader_workbench/domains/plate_reader/ordering.py +59 -0
- reader_workbench/domains/plate_reader/plots/__init__.py +15 -0
- reader_workbench/domains/plate_reader/plots/_data.py +29 -0
- reader_workbench/domains/plate_reader/plots/common.py +346 -0
- reader_workbench/domains/plate_reader/plots/distributions.py +324 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych.py +525 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych_render.py +195 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/__init__.py +26 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic.py +295 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_components.py +149 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_render.py +308 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_style.py +51 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/schema.py +8 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/summary.py +140 -0
- reader_workbench/domains/plate_reader/plots/grouping.py +53 -0
- reader_workbench/domains/plate_reader/plots/panels/__init__.py +12 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot.py +161 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot_data.py +91 -0
- reader_workbench/domains/plate_reader/plots/panels/time_series.py +296 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic.py +435 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic_render.py +300 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/__init__.py +315 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/planning.py +168 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/__init__.py +205 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/inputs.py +103 -0
- reader_workbench/domains/plate_reader/plots/time_series.py +317 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/__init__.py +479 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/planning.py +283 -0
- reader_workbench/domains/time_series/__init__.py +29 -0
- reader_workbench/domains/time_series/aggregation.py +60 -0
- reader_workbench/domains/time_series/contracts.py +368 -0
- reader_workbench/domains/time_series/reduction.py +395 -0
- reader_workbench/errors.py +55 -0
- reader_workbench/maintenance/__init__.py +6 -0
- reader_workbench/maintenance/docs.py +335 -0
- reader_workbench/maintenance/model.py +28 -0
- reader_workbench/maintenance/release.py +39 -0
- reader_workbench/maintenance/skills.py +124 -0
- reader_workbench/plotting/__init__.py +20 -0
- reader_workbench/plotting/mpl.py +56 -0
- reader_workbench/plotting/sinks.py +69 -0
- reader_workbench/plotting/style.py +175 -0
- reader_workbench/plotting/utils.py +27 -0
- reader_workbench/plugins/__init__.py +1 -0
- reader_workbench/plugins/catalog.py +33 -0
- reader_workbench/plugins/export/__init__.py +0 -0
- reader_workbench/plugins/export/_paths.py +21 -0
- reader_workbench/plugins/export/csv.py +41 -0
- reader_workbench/plugins/export/xlsx.py +44 -0
- reader_workbench/plugins/ingest/__init__.py +0 -0
- reader_workbench/plugins/ingest/_discovery.py +58 -0
- reader_workbench/plugins/ingest/discovery_policy.py +66 -0
- reader_workbench/plugins/ingest/flow_cytometer.py +139 -0
- reader_workbench/plugins/ingest/synergy_h1.py +234 -0
- reader_workbench/plugins/manifests/__init__.py +1 -0
- reader_workbench/plugins/manifests/export.py +29 -0
- reader_workbench/plugins/manifests/ingest.py +29 -0
- reader_workbench/plugins/manifests/plot.py +161 -0
- reader_workbench/plugins/manifests/transform.py +172 -0
- reader_workbench/plugins/manifests/validator.py +18 -0
- reader_workbench/plugins/plot/__init__.py +0 -0
- reader_workbench/plugins/plot/_shared.py +55 -0
- reader_workbench/plugins/plot/cytometry_diagnostic.py +54 -0
- reader_workbench/plugins/plot/distributions.py +58 -0
- reader_workbench/plugins/plot/dual_reporter_triptych.py +224 -0
- reader_workbench/plugins/plot/four_state_event_window_diagnostic.py +133 -0
- reader_workbench/plugins/plot/four_state_event_window_summary.py +53 -0
- reader_workbench/plugins/plot/four_state_vector_collection.py +45 -0
- reader_workbench/plugins/plot/four_state_vector_diagnostic.py +105 -0
- reader_workbench/plugins/plot/four_state_vector_heatmap.py +63 -0
- reader_workbench/plugins/plot/logic_symmetry.py +56 -0
- reader_workbench/plugins/plot/single_reporter_diagnostic.py +279 -0
- reader_workbench/plugins/plot/snapshot_barplot.py +65 -0
- reader_workbench/plugins/plot/snapshot_heatmap.py +104 -0
- reader_workbench/plugins/plot/time_series.py +114 -0
- reader_workbench/plugins/plot/ts_and_snap.py +210 -0
- reader_workbench/plugins/transform/__init__.py +0 -0
- reader_workbench/plugins/transform/_four_state_vector.py +204 -0
- reader_workbench/plugins/transform/_labeling.py +109 -0
- reader_workbench/plugins/transform/alias.py +70 -0
- reader_workbench/plugins/transform/assay_labels.py +62 -0
- reader_workbench/plugins/transform/blank.py +79 -0
- reader_workbench/plugins/transform/crosstalk_pairs.py +180 -0
- reader_workbench/plugins/transform/cytometry_gating.py +120 -0
- reader_workbench/plugins/transform/fold_change.py +79 -0
- reader_workbench/plugins/transform/four_state_event_window.py +93 -0
- reader_workbench/plugins/transform/four_state_vector.py +62 -0
- reader_workbench/plugins/transform/four_state_vector_collection.py +41 -0
- reader_workbench/plugins/transform/logic_symmetry.py +67 -0
- reader_workbench/plugins/transform/outlier_filter.py +60 -0
- reader_workbench/plugins/transform/overflow.py +197 -0
- reader_workbench/plugins/transform/ratio.py +237 -0
- reader_workbench/plugins/transform/sample_map.py +170 -0
- reader_workbench/plugins/transform/sample_metadata.py +94 -0
- reader_workbench/plugins/validator/__init__.py +1 -0
- reader_workbench/plugins/validator/to_tidy_plus_map.py +155 -0
- reader_workbench/protocols/__init__.py +80 -0
- reader_workbench/protocols/_builtins_plate_reader_growth.py +179 -0
- reader_workbench/protocols/_builtins_plate_reader_variants.py +274 -0
- reader_workbench/protocols/builtins.py +1656 -0
- reader_workbench/protocols/compiler.py +22 -0
- reader_workbench/protocols/compilers/__init__.py +1 -0
- reader_workbench/protocols/compilers/common.py +100 -0
- reader_workbench/protocols/compilers/cytometry.py +87 -0
- reader_workbench/protocols/compilers/generic.py +14 -0
- reader_workbench/protocols/compilers/logic.py +245 -0
- reader_workbench/protocols/compilers/plate_reader.py +937 -0
- reader_workbench/protocols/compilers/plate_reader_pipeline.py +197 -0
- reader_workbench/protocols/model.py +1486 -0
- reader_workbench/protocols/semantic_coverage.py +234 -0
- reader_workbench/runtime/__init__.py +12 -0
- reader_workbench/runtime/builtin.py +23 -0
- reader_workbench/runtime/model.py +42 -0
- reader_workbench/workbench/__init__.py +60 -0
- reader_workbench/workbench/assets/__init__.py +22 -0
- reader_workbench/workbench/assets/types.py +118 -0
- reader_workbench/workbench/audit/__init__.py +5 -0
- reader_workbench/workbench/audit/experiments.py +307 -0
- reader_workbench/workbench/audit/staging.py +187 -0
- reader_workbench/workbench/cli/__init__.py +51 -0
- reader_workbench/workbench/cli/_lazy.py +9 -0
- reader_workbench/workbench/cli/_records_view.py +150 -0
- reader_workbench/workbench/cli/_surface_execution.py +443 -0
- reader_workbench/workbench/cli/audit.py +95 -0
- reader_workbench/workbench/cli/automation.py +229 -0
- reader_workbench/workbench/cli/demo.py +46 -0
- reader_workbench/workbench/cli/dop.py +91 -0
- reader_workbench/workbench/cli/experiments.py +635 -0
- reader_workbench/workbench/cli/helpers.py +232 -0
- reader_workbench/workbench/cli/main.py +59 -0
- reader_workbench/workbench/cli/maintenance.py +82 -0
- reader_workbench/workbench/cli/notebooks.py +260 -0
- reader_workbench/workbench/cli/pagination.py +117 -0
- reader_workbench/workbench/cli/protocols.py +336 -0
- reader_workbench/workbench/cli/shared.py +309 -0
- reader_workbench/workbench/cli/surfaces.py +534 -0
- reader_workbench/workbench/cli/verification.py +128 -0
- reader_workbench/workbench/commands.py +10 -0
- reader_workbench/workbench/config/__init__.py +47 -0
- reader_workbench/workbench/config/identity.py +13 -0
- reader_workbench/workbench/config/load.py +405 -0
- reader_workbench/workbench/config/model.py +274 -0
- reader_workbench/workbench/context.py +26 -0
- reader_workbench/workbench/decl/__init__.py +31 -0
- reader_workbench/workbench/decl/build.py +190 -0
- reader_workbench/workbench/decl/model.py +81 -0
- reader_workbench/workbench/dop/__init__.py +12 -0
- reader_workbench/workbench/dop/builtins.py +261 -0
- reader_workbench/workbench/dop/model.py +209 -0
- reader_workbench/workbench/engine/__init__.py +42 -0
- reader_workbench/workbench/engine/_shared.py +76 -0
- reader_workbench/workbench/engine/contracts.py +283 -0
- reader_workbench/workbench/engine/execution.py +326 -0
- reader_workbench/workbench/engine/file_outputs.py +260 -0
- reader_workbench/workbench/engine/inputs.py +161 -0
- reader_workbench/workbench/engine/invocations.py +507 -0
- reader_workbench/workbench/engine/planning.py +72 -0
- reader_workbench/workbench/engine/runtime.py +464 -0
- reader_workbench/workbench/engine/setup.py +149 -0
- reader_workbench/workbench/engine/validation.py +684 -0
- reader_workbench/workbench/experiment/__init__.py +47 -0
- reader_workbench/workbench/experiment/model.py +381 -0
- reader_workbench/workbench/experiments.py +133 -0
- reader_workbench/workbench/graph/__init__.py +47 -0
- reader_workbench/workbench/graph/nodes.py +102 -0
- reader_workbench/workbench/graph/normalize.py +177 -0
- reader_workbench/workbench/graph/refs.py +148 -0
- reader_workbench/workbench/input_discovery.py +19 -0
- reader_workbench/workbench/inspection/__init__.py +3 -0
- reader_workbench/workbench/inspection/catalogs.py +128 -0
- reader_workbench/workbench/inspection/common.py +92 -0
- reader_workbench/workbench/inspection/dop.py +64 -0
- reader_workbench/workbench/inspection/experiments.py +449 -0
- reader_workbench/workbench/inspection/inventory.py +68 -0
- reader_workbench/workbench/inspection/protocols.py +368 -0
- reader_workbench/workbench/inspection/readiness.py +333 -0
- reader_workbench/workbench/inspection/reports.py +367 -0
- reader_workbench/workbench/inspection/results.py +166 -0
- reader_workbench/workbench/inspection/runtime.py +287 -0
- reader_workbench/workbench/inspection/semantics.py +192 -0
- reader_workbench/workbench/inspection/validation.py +30 -0
- reader_workbench/workbench/notebooks/__init__.py +17 -0
- reader_workbench/workbench/notebooks/_launch_registry.py +112 -0
- reader_workbench/workbench/notebooks/_launch_runtime.py +104 -0
- reader_workbench/workbench/notebooks/components/__init__.py +21 -0
- reader_workbench/workbench/notebooks/components/deliverables.py +403 -0
- reader_workbench/workbench/notebooks/components/overview.py +119 -0
- reader_workbench/workbench/notebooks/eda.marimo.py.txt +153 -0
- reader_workbench/workbench/notebooks/launch.py +274 -0
- reader_workbench/workbench/notebooks/presentation.py +136 -0
- reader_workbench/workbench/notebooks/scaffold.py +60 -0
- reader_workbench/workbench/ontology.py +78 -0
- reader_workbench/workbench/paths.py +44 -0
- reader_workbench/workbench/ports/__init__.py +31 -0
- reader_workbench/workbench/ports/model.py +168 -0
- reader_workbench/workbench/records/__init__.py +44 -0
- reader_workbench/workbench/records/epoch.py +329 -0
- reader_workbench/workbench/records/evidence.py +247 -0
- reader_workbench/workbench/records/identity.py +87 -0
- reader_workbench/workbench/records/locking.py +185 -0
- reader_workbench/workbench/records/model.py +711 -0
- reader_workbench/workbench/records/sources.py +73 -0
- reader_workbench/workbench/records/store.py +1022 -0
- reader_workbench/workbench/records/verification.py +998 -0
- reader_workbench/workbench/registry.py +333 -0
- reader_workbench/workbench/spec_overrides.py +215 -0
- reader_workbench-1.0.0.dist-info/METADATA +91 -0
- reader_workbench-1.0.0.dist-info/RECORD +293 -0
- reader_workbench-1.0.0.dist-info/WHEEL +5 -0
- reader_workbench-1.0.0.dist-info/entry_points.txt +2 -0
- reader_workbench-1.0.0.dist-info/licenses/LICENSE +21 -0
- reader_workbench-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
"""Sequential gating and statistical summaries for cytometry events."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import math
|
|
6
|
+
|
|
7
|
+
import polars as pl
|
|
8
|
+
|
|
9
|
+
from reader_workbench.domains.cytometry.analysis.events import (
|
|
10
|
+
_METADATA_COLUMNS,
|
|
11
|
+
CytometryAnalysis,
|
|
12
|
+
CytometryAnalysisError,
|
|
13
|
+
GateSpec,
|
|
14
|
+
ThresholdSpec,
|
|
15
|
+
_require_columns,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def analyze_events(
|
|
20
|
+
event_table: pl.DataFrame,
|
|
21
|
+
*,
|
|
22
|
+
gate: GateSpec,
|
|
23
|
+
threshold: ThresholdSpec,
|
|
24
|
+
group_column: str | None = None,
|
|
25
|
+
) -> CytometryAnalysis:
|
|
26
|
+
"""Apply sequential gates and compute per-sample and group summaries in Polars."""
|
|
27
|
+
|
|
28
|
+
required_columns = [threshold.channel, "sample_id"]
|
|
29
|
+
if gate.cells_enabled:
|
|
30
|
+
required_columns.extend((gate.cells_x_channel, gate.cells_y_channel))
|
|
31
|
+
if gate.singlets_enabled:
|
|
32
|
+
required_columns.extend((gate.singlet_x_channel, gate.singlet_y_channel))
|
|
33
|
+
_require_columns(event_table, required_columns)
|
|
34
|
+
|
|
35
|
+
cells_mask = pl.lit(True)
|
|
36
|
+
if gate.cells_enabled:
|
|
37
|
+
_validate_interval("cells X", gate.cells_x_range)
|
|
38
|
+
_validate_interval("cells Y", gate.cells_y_range)
|
|
39
|
+
cells_x = pl.col(gate.cells_x_channel).cast(pl.Float64, strict=False)
|
|
40
|
+
cells_y = pl.col(gate.cells_y_channel).cast(pl.Float64, strict=False)
|
|
41
|
+
cells_mask = (
|
|
42
|
+
cells_x.is_finite()
|
|
43
|
+
& cells_y.is_finite()
|
|
44
|
+
& cells_x.is_between(*gate.cells_x_range, closed="both")
|
|
45
|
+
& cells_y.is_between(*gate.cells_y_range, closed="both")
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
singlet_mask = pl.lit(True)
|
|
49
|
+
if gate.singlets_enabled:
|
|
50
|
+
_validate_interval("singlet ratio", gate.singlet_ratio_range)
|
|
51
|
+
singlet_x = pl.col(gate.singlet_x_channel).cast(pl.Float64, strict=False)
|
|
52
|
+
singlet_y = pl.col(gate.singlet_y_channel).cast(pl.Float64, strict=False)
|
|
53
|
+
ratio = singlet_y / singlet_x
|
|
54
|
+
singlet_mask = ratio.is_finite() & ratio.is_between(*gate.singlet_ratio_range, closed="both")
|
|
55
|
+
|
|
56
|
+
cells_mask_column = "__reader_cells_mask"
|
|
57
|
+
gate_mask_column = "__reader_gate_mask"
|
|
58
|
+
work = event_table.with_columns(
|
|
59
|
+
cells_mask.alias(cells_mask_column),
|
|
60
|
+
(cells_mask & singlet_mask).alias(gate_mask_column),
|
|
61
|
+
)
|
|
62
|
+
gated_events = work.filter(pl.col(gate_mask_column)).drop(cells_mask_column, gate_mask_column)
|
|
63
|
+
if gated_events.is_empty():
|
|
64
|
+
raise CytometryAnalysisError("No events remain after gating. Adjust ranges.")
|
|
65
|
+
|
|
66
|
+
if group_column is not None:
|
|
67
|
+
if not isinstance(group_column, str) or not group_column.strip():
|
|
68
|
+
raise CytometryAnalysisError("Group column must be a non-empty string or null.")
|
|
69
|
+
group_column = group_column.strip()
|
|
70
|
+
_require_columns(event_table, (group_column,))
|
|
71
|
+
metadata_columns = list(
|
|
72
|
+
dict.fromkeys(
|
|
73
|
+
(
|
|
74
|
+
*[column for column in _METADATA_COLUMNS if column in event_table.columns],
|
|
75
|
+
*([group_column] if group_column else []),
|
|
76
|
+
)
|
|
77
|
+
)
|
|
78
|
+
)
|
|
79
|
+
counts = work.group_by("sample_id", maintain_order=True).agg(
|
|
80
|
+
*[pl.col(column).first().alias(column) for column in metadata_columns],
|
|
81
|
+
pl.len().alias("n_total_events"),
|
|
82
|
+
pl.col(cells_mask_column).sum().cast(pl.Int64).alias("n_cells_gate"),
|
|
83
|
+
pl.col(gate_mask_column).sum().cast(pl.Int64).alias("n_singlets"),
|
|
84
|
+
)
|
|
85
|
+
counts = counts.with_columns(
|
|
86
|
+
pl.when(pl.col("n_total_events") > 0)
|
|
87
|
+
.then(100.0 * pl.col("n_cells_gate") / pl.col("n_total_events"))
|
|
88
|
+
.otherwise(float("nan"))
|
|
89
|
+
.alias("pct_cells"),
|
|
90
|
+
pl.when(pl.col("n_cells_gate") > 0)
|
|
91
|
+
.then(100.0 * pl.col("n_singlets") / pl.col("n_cells_gate"))
|
|
92
|
+
.otherwise(float("nan"))
|
|
93
|
+
.alias("pct_singlets_of_cells"),
|
|
94
|
+
pl.when(pl.col("n_total_events") > 0)
|
|
95
|
+
.then(100.0 * pl.col("n_singlets") / pl.col("n_total_events"))
|
|
96
|
+
.otherwise(float("nan"))
|
|
97
|
+
.alias("pct_final"),
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
threshold_value = _resolve_threshold(gated_events, threshold)
|
|
101
|
+
fluor_column = "__reader_fluor"
|
|
102
|
+
gated_for_stats = gated_events.with_columns(
|
|
103
|
+
pl.col(threshold.channel).cast(pl.Float64, strict=False).alias(fluor_column)
|
|
104
|
+
)
|
|
105
|
+
finite_fluor = pl.col(fluor_column).filter(pl.col(fluor_column).is_finite())
|
|
106
|
+
positive_fluor = finite_fluor.filter(finite_fluor > 0)
|
|
107
|
+
sample_stats = gated_for_stats.group_by("sample_id", maintain_order=True).agg(
|
|
108
|
+
finite_fluor.median().alias("fluor_median"),
|
|
109
|
+
finite_fluor.mean().alias("fluor_mean"),
|
|
110
|
+
positive_fluor.log().mean().exp().alias("fluor_geomean"),
|
|
111
|
+
finite_fluor.quantile(0.90, interpolation="linear").alias("fluor_p90"),
|
|
112
|
+
finite_fluor.quantile(0.99, interpolation="linear").alias("fluor_p99"),
|
|
113
|
+
(100.0 * (finite_fluor > threshold_value).mean()).alias("pct_positive"),
|
|
114
|
+
)
|
|
115
|
+
stats_sample = counts.join(sample_stats, on="sample_id", how="left")
|
|
116
|
+
|
|
117
|
+
stats_group = None
|
|
118
|
+
if group_column is not None:
|
|
119
|
+
stats_group = stats_sample.group_by(group_column, maintain_order=True).agg(
|
|
120
|
+
pl.col("sample_id").n_unique().alias("n_samples"),
|
|
121
|
+
pl.col("fluor_median").mean().alias("fluor_median_mean"),
|
|
122
|
+
pl.col("fluor_median").std().alias("fluor_median_std"),
|
|
123
|
+
pl.col("fluor_geomean").mean().alias("fluor_geomean_mean"),
|
|
124
|
+
pl.col("pct_positive").mean().alias("pct_positive_mean"),
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
all_fluor = pl.col(threshold.channel).cast(pl.Float64, strict=False)
|
|
128
|
+
finite_all_fluor = all_fluor.filter(all_fluor.is_finite())
|
|
129
|
+
qc_table = event_table.group_by("sample_id", maintain_order=True).agg(
|
|
130
|
+
pl.when(finite_all_fluor.len() > 0)
|
|
131
|
+
.then(100.0 * (finite_all_fluor <= 0).mean())
|
|
132
|
+
.otherwise(float("nan"))
|
|
133
|
+
.alias("pct_nonpositive")
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
return CytometryAnalysis(
|
|
137
|
+
gated_events=gated_events,
|
|
138
|
+
gate_counts_sample=counts,
|
|
139
|
+
stats_sample=stats_sample,
|
|
140
|
+
stats_group=stats_group,
|
|
141
|
+
qc_table=qc_table,
|
|
142
|
+
threshold_value=threshold_value,
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _validate_interval(label: str, interval: tuple[float, float]) -> None:
|
|
147
|
+
low, high = interval
|
|
148
|
+
if not math.isfinite(low) or not math.isfinite(high) or high < low:
|
|
149
|
+
raise CytometryAnalysisError(f"{label} range must contain two finite values in ascending order.")
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _resolve_threshold(gated_events: pl.DataFrame, threshold: ThresholdSpec) -> float:
|
|
153
|
+
if threshold.mode == "manual":
|
|
154
|
+
value = float(threshold.value)
|
|
155
|
+
elif threshold.mode == "from_control_quantile":
|
|
156
|
+
if threshold.group_column is None or threshold.control_value is None:
|
|
157
|
+
raise CytometryAnalysisError("Control thresholding requires a group column and control value.")
|
|
158
|
+
_require_columns(gated_events, (threshold.group_column, threshold.channel))
|
|
159
|
+
if not 0.0 <= threshold.quantile <= 1.0:
|
|
160
|
+
raise CytometryAnalysisError("Control quantile must be between 0 and 1.")
|
|
161
|
+
control_values = (
|
|
162
|
+
gated_events.filter(pl.col(threshold.group_column).cast(pl.String) == threshold.control_value)
|
|
163
|
+
.get_column(threshold.channel)
|
|
164
|
+
.cast(pl.Float64, strict=False)
|
|
165
|
+
.drop_nulls()
|
|
166
|
+
)
|
|
167
|
+
control_values = control_values.filter(control_values.is_finite())
|
|
168
|
+
if control_values.is_empty():
|
|
169
|
+
raise CytometryAnalysisError("No control events are available for thresholding.")
|
|
170
|
+
value = float(control_values.quantile(threshold.quantile, interpolation="linear"))
|
|
171
|
+
else:
|
|
172
|
+
raise CytometryAnalysisError(f"Unknown threshold mode `{threshold.mode}`.")
|
|
173
|
+
if not math.isfinite(value):
|
|
174
|
+
raise CytometryAnalysisError("Threshold value must be finite.")
|
|
175
|
+
return value
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
"""Explicit normal-lifecycle cytometry gating workflow."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import Literal
|
|
7
|
+
|
|
8
|
+
import pandas as pd
|
|
9
|
+
import polars as pl
|
|
10
|
+
|
|
11
|
+
from .events import CytometryAnalysisError, GateSpec, ThresholdSpec, prepare_event_table
|
|
12
|
+
from .gating import analyze_events
|
|
13
|
+
|
|
14
|
+
_NONPOSITIVE_EVALUABLE_EVENTS_COLUMN = "__reader_nonpositive_evaluable_events"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True, slots=True)
|
|
18
|
+
class CytometryQCSpec:
|
|
19
|
+
minimum_final_events: int
|
|
20
|
+
minimum_final_percent: float
|
|
21
|
+
maximum_nonpositive_percent: float
|
|
22
|
+
nonpositive_scope: Literal["all_events", "gated_events"]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(frozen=True, slots=True)
|
|
26
|
+
class CytometryGatingRequest:
|
|
27
|
+
gate: GateSpec
|
|
28
|
+
threshold: ThresholdSpec
|
|
29
|
+
group_column: str | None
|
|
30
|
+
qc: CytometryQCSpec
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True, slots=True)
|
|
34
|
+
class CytometryGatingResult:
|
|
35
|
+
gate_definition: pl.DataFrame
|
|
36
|
+
gated_events: pl.DataFrame
|
|
37
|
+
sample_stats: pl.DataFrame
|
|
38
|
+
group_stats: pl.DataFrame
|
|
39
|
+
qc: pl.DataFrame
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def run_cytometry_gating(events: pd.DataFrame | pl.DataFrame, request: CytometryGatingRequest) -> CytometryGatingResult:
|
|
43
|
+
"""Resolve an explicit gating request into typed, persistence-ready tables."""
|
|
44
|
+
|
|
45
|
+
_validate_request(request)
|
|
46
|
+
source = pl.from_pandas(events) if isinstance(events, pd.DataFrame) else events
|
|
47
|
+
if not isinstance(source, pl.DataFrame):
|
|
48
|
+
raise TypeError(f"Expected pandas or Polars event data, got {type(events).__name__}.")
|
|
49
|
+
|
|
50
|
+
selected_channels: list[str] = []
|
|
51
|
+
if request.gate.cells_enabled:
|
|
52
|
+
selected_channels.extend((request.gate.cells_x_channel, request.gate.cells_y_channel))
|
|
53
|
+
if request.gate.singlets_enabled:
|
|
54
|
+
selected_channels.extend((request.gate.singlet_x_channel, request.gate.singlet_y_channel))
|
|
55
|
+
selected_channels.append(request.threshold.channel)
|
|
56
|
+
metadata_columns = tuple(
|
|
57
|
+
dict.fromkeys(
|
|
58
|
+
column
|
|
59
|
+
for column in (request.group_column, request.threshold.group_column)
|
|
60
|
+
if isinstance(column, str) and column
|
|
61
|
+
)
|
|
62
|
+
)
|
|
63
|
+
wide = prepare_event_table(
|
|
64
|
+
source,
|
|
65
|
+
channels=tuple(dict.fromkeys(selected_channels)),
|
|
66
|
+
metadata_columns=metadata_columns,
|
|
67
|
+
)
|
|
68
|
+
analysis = analyze_events(
|
|
69
|
+
wide,
|
|
70
|
+
gate=request.gate,
|
|
71
|
+
threshold=request.threshold,
|
|
72
|
+
group_column=request.group_column,
|
|
73
|
+
)
|
|
74
|
+
sample_stats = _sample_stats(analysis.stats_sample, request=request, threshold_value=analysis.threshold_value)
|
|
75
|
+
group_stats = _group_stats(analysis.stats_group, request=request)
|
|
76
|
+
nonpositive_source = wide if request.qc.nonpositive_scope == "all_events" else analysis.gated_events
|
|
77
|
+
qc = _qc_table(
|
|
78
|
+
analysis.gate_counts_sample,
|
|
79
|
+
_nonpositive_table(nonpositive_source, channel=request.threshold.channel),
|
|
80
|
+
request=request,
|
|
81
|
+
)
|
|
82
|
+
return CytometryGatingResult(
|
|
83
|
+
gate_definition=_gate_definition(request, threshold_value=analysis.threshold_value),
|
|
84
|
+
gated_events=analysis.gated_events,
|
|
85
|
+
sample_stats=sample_stats,
|
|
86
|
+
group_stats=group_stats,
|
|
87
|
+
qc=qc,
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _validate_request(request: CytometryGatingRequest) -> None:
|
|
92
|
+
if request.group_column is not None and (
|
|
93
|
+
not isinstance(request.group_column, str) or not request.group_column.strip()
|
|
94
|
+
):
|
|
95
|
+
raise CytometryAnalysisError("group_column must be a non-empty string or null.")
|
|
96
|
+
if request.threshold.mode == "manual":
|
|
97
|
+
if request.threshold.group_column is not None or request.threshold.control_value is not None:
|
|
98
|
+
raise CytometryAnalysisError("Manual thresholding may not declare control-group fields.")
|
|
99
|
+
elif request.threshold.mode == "from_control_quantile":
|
|
100
|
+
if not request.threshold.group_column or not request.threshold.control_value:
|
|
101
|
+
raise CytometryAnalysisError("Control thresholding requires explicit group_column and control_value.")
|
|
102
|
+
else:
|
|
103
|
+
raise CytometryAnalysisError(f"Unknown threshold mode `{request.threshold.mode}`.")
|
|
104
|
+
if request.qc.minimum_final_events < 0:
|
|
105
|
+
raise CytometryAnalysisError("minimum_final_events must be nonnegative.")
|
|
106
|
+
if request.qc.nonpositive_scope not in {"all_events", "gated_events"}:
|
|
107
|
+
raise CytometryAnalysisError("nonpositive_scope must be 'all_events' or 'gated_events'.")
|
|
108
|
+
for name, value in (
|
|
109
|
+
("minimum_final_percent", request.qc.minimum_final_percent),
|
|
110
|
+
("maximum_nonpositive_percent", request.qc.maximum_nonpositive_percent),
|
|
111
|
+
):
|
|
112
|
+
if not 0.0 <= float(value) <= 100.0:
|
|
113
|
+
raise CytometryAnalysisError(f"{name} must be between 0 and 100.")
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _gate_definition(request: CytometryGatingRequest, *, threshold_value: float) -> pl.DataFrame:
|
|
117
|
+
return pl.DataFrame(
|
|
118
|
+
{
|
|
119
|
+
"definition_id": ["resolved"],
|
|
120
|
+
"cells_enabled": [request.gate.cells_enabled],
|
|
121
|
+
"cells_x_channel": [request.gate.cells_x_channel],
|
|
122
|
+
"cells_x_min": [float(request.gate.cells_x_range[0])],
|
|
123
|
+
"cells_x_max": [float(request.gate.cells_x_range[1])],
|
|
124
|
+
"cells_y_channel": [request.gate.cells_y_channel],
|
|
125
|
+
"cells_y_min": [float(request.gate.cells_y_range[0])],
|
|
126
|
+
"cells_y_max": [float(request.gate.cells_y_range[1])],
|
|
127
|
+
"singlets_enabled": [request.gate.singlets_enabled],
|
|
128
|
+
"singlet_x_channel": [request.gate.singlet_x_channel],
|
|
129
|
+
"singlet_y_channel": [request.gate.singlet_y_channel],
|
|
130
|
+
"singlet_ratio_min": [float(request.gate.singlet_ratio_range[0])],
|
|
131
|
+
"singlet_ratio_max": [float(request.gate.singlet_ratio_range[1])],
|
|
132
|
+
"fluorescence_channel": [request.threshold.channel],
|
|
133
|
+
"threshold_mode": [request.threshold.mode],
|
|
134
|
+
"threshold_value": [float(threshold_value)],
|
|
135
|
+
"threshold_group_column": [request.threshold.group_column],
|
|
136
|
+
"threshold_control_value": [request.threshold.control_value],
|
|
137
|
+
"threshold_quantile": [
|
|
138
|
+
float(request.threshold.quantile) if request.threshold.mode == "from_control_quantile" else None
|
|
139
|
+
],
|
|
140
|
+
"group_column": [request.group_column],
|
|
141
|
+
"minimum_final_events": [int(request.qc.minimum_final_events)],
|
|
142
|
+
"minimum_final_percent": [float(request.qc.minimum_final_percent)],
|
|
143
|
+
"maximum_nonpositive_percent": [float(request.qc.maximum_nonpositive_percent)],
|
|
144
|
+
"nonpositive_scope": [request.qc.nonpositive_scope],
|
|
145
|
+
},
|
|
146
|
+
schema_overrides={
|
|
147
|
+
"threshold_group_column": pl.String,
|
|
148
|
+
"threshold_control_value": pl.String,
|
|
149
|
+
"threshold_quantile": pl.Float64,
|
|
150
|
+
"group_column": pl.String,
|
|
151
|
+
},
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _sample_stats(stats: pl.DataFrame, *, request: CytometryGatingRequest, threshold_value: float) -> pl.DataFrame:
|
|
156
|
+
group_value = (
|
|
157
|
+
pl.col(request.group_column).cast(pl.String)
|
|
158
|
+
if request.group_column is not None
|
|
159
|
+
else pl.lit(None, dtype=pl.String)
|
|
160
|
+
)
|
|
161
|
+
return stats.select(
|
|
162
|
+
"sample_id",
|
|
163
|
+
pl.lit(request.group_column, dtype=pl.String).alias("group_column"),
|
|
164
|
+
group_value.alias("group_value"),
|
|
165
|
+
"n_total_events",
|
|
166
|
+
"n_cells_gate",
|
|
167
|
+
"n_singlets",
|
|
168
|
+
"pct_cells",
|
|
169
|
+
"pct_singlets_of_cells",
|
|
170
|
+
"pct_final",
|
|
171
|
+
"fluor_median",
|
|
172
|
+
"fluor_mean",
|
|
173
|
+
"fluor_geomean",
|
|
174
|
+
"fluor_p90",
|
|
175
|
+
"fluor_p99",
|
|
176
|
+
"pct_positive",
|
|
177
|
+
pl.lit(request.threshold.channel).alias("fluorescence_channel"),
|
|
178
|
+
pl.lit(float(threshold_value)).alias("threshold_value"),
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _group_stats(stats: pl.DataFrame | None, *, request: CytometryGatingRequest) -> pl.DataFrame:
|
|
183
|
+
schema = {
|
|
184
|
+
"group_column": pl.String,
|
|
185
|
+
"group_value": pl.String,
|
|
186
|
+
"n_samples": pl.Int64,
|
|
187
|
+
"fluor_median_mean": pl.Float64,
|
|
188
|
+
"fluor_median_std": pl.Float64,
|
|
189
|
+
"fluor_geomean_mean": pl.Float64,
|
|
190
|
+
"pct_positive_mean": pl.Float64,
|
|
191
|
+
}
|
|
192
|
+
if request.group_column is None or stats is None:
|
|
193
|
+
return pl.DataFrame(schema=schema)
|
|
194
|
+
return stats.select(
|
|
195
|
+
pl.lit(request.group_column).alias("group_column"),
|
|
196
|
+
pl.col(request.group_column).cast(pl.String).alias("group_value"),
|
|
197
|
+
"n_samples",
|
|
198
|
+
"fluor_median_mean",
|
|
199
|
+
"fluor_median_std",
|
|
200
|
+
"fluor_geomean_mean",
|
|
201
|
+
"pct_positive_mean",
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _qc_table(counts: pl.DataFrame, nonpositive: pl.DataFrame, *, request: CytometryGatingRequest) -> pl.DataFrame:
|
|
206
|
+
joined = counts.select(
|
|
207
|
+
"sample_id",
|
|
208
|
+
"n_total_events",
|
|
209
|
+
"n_cells_gate",
|
|
210
|
+
"n_singlets",
|
|
211
|
+
"pct_final",
|
|
212
|
+
).join(nonpositive, on="sample_id", how="left")
|
|
213
|
+
# Keep the persisted percentage finite, but require a real denominator so
|
|
214
|
+
# even a permissive 100% ceiling cannot pass an unevaluable sample.
|
|
215
|
+
return (
|
|
216
|
+
joined.with_columns(
|
|
217
|
+
pl.col("pct_nonpositive").fill_nan(100.0).fill_null(100.0),
|
|
218
|
+
pl.col(_NONPOSITIVE_EVALUABLE_EVENTS_COLUMN).fill_null(0).cast(pl.Int64),
|
|
219
|
+
pl.lit(int(request.qc.minimum_final_events)).alias("minimum_final_events"),
|
|
220
|
+
pl.lit(float(request.qc.minimum_final_percent)).alias("minimum_final_percent"),
|
|
221
|
+
pl.lit(float(request.qc.maximum_nonpositive_percent)).alias("maximum_nonpositive_percent"),
|
|
222
|
+
pl.lit(request.qc.nonpositive_scope).alias("nonpositive_scope"),
|
|
223
|
+
)
|
|
224
|
+
.with_columns(
|
|
225
|
+
(pl.col("n_singlets") >= pl.col("minimum_final_events")).alias("passes_final_events"),
|
|
226
|
+
(pl.col("pct_final") >= pl.col("minimum_final_percent")).alias("passes_final_percent"),
|
|
227
|
+
(
|
|
228
|
+
(pl.col(_NONPOSITIVE_EVALUABLE_EVENTS_COLUMN) > 0)
|
|
229
|
+
& (pl.col("pct_nonpositive") <= pl.col("maximum_nonpositive_percent"))
|
|
230
|
+
).alias("passes_nonpositive"),
|
|
231
|
+
)
|
|
232
|
+
.with_columns(
|
|
233
|
+
(pl.col("passes_final_events") & pl.col("passes_final_percent") & pl.col("passes_nonpositive")).alias(
|
|
234
|
+
"qc_pass"
|
|
235
|
+
)
|
|
236
|
+
)
|
|
237
|
+
.with_columns(pl.when(pl.col("qc_pass")).then(pl.lit("pass")).otherwise(pl.lit("fail")).alias("qc_status"))
|
|
238
|
+
.drop(_NONPOSITIVE_EVALUABLE_EVENTS_COLUMN)
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def _nonpositive_table(events: pl.DataFrame, *, channel: str) -> pl.DataFrame:
|
|
243
|
+
fluorescence = pl.col(channel).cast(pl.Float64, strict=False)
|
|
244
|
+
finite = fluorescence.filter(fluorescence.is_finite())
|
|
245
|
+
return events.group_by("sample_id", maintain_order=True).agg(
|
|
246
|
+
finite.len().cast(pl.Int64).alias(_NONPOSITIVE_EVALUABLE_EVENTS_COLUMN),
|
|
247
|
+
pl.when(finite.len() > 0).then(100.0 * (finite <= 0).mean()).otherwise(100.0).alias("pct_nonpositive"),
|
|
248
|
+
)
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""FCS parsing for cytometry experiments."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Literal
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
import pandas as pd
|
|
11
|
+
|
|
12
|
+
from reader_workbench.errors import ParseError
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _to_float(value) -> float:
|
|
16
|
+
if value is None:
|
|
17
|
+
return float("nan")
|
|
18
|
+
try:
|
|
19
|
+
return float(value)
|
|
20
|
+
except Exception:
|
|
21
|
+
return float("nan")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _parse_pne(value) -> tuple[float, float]:
|
|
25
|
+
if value is None:
|
|
26
|
+
return (float("nan"), float("nan"))
|
|
27
|
+
if isinstance(value, (tuple, list)) and len(value) == 2:
|
|
28
|
+
return (_to_float(value[0]), _to_float(value[1]))
|
|
29
|
+
text = str(value).strip()
|
|
30
|
+
if "," in text:
|
|
31
|
+
left, right = text.split(",", 1)
|
|
32
|
+
return (_to_float(left.strip()), _to_float(right.strip()))
|
|
33
|
+
return (float("nan"), float("nan"))
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _clean_text(value) -> str | None:
|
|
37
|
+
if value is None:
|
|
38
|
+
return None
|
|
39
|
+
if isinstance(value, bytes):
|
|
40
|
+
try:
|
|
41
|
+
return value.decode("utf-8", errors="ignore")
|
|
42
|
+
except Exception:
|
|
43
|
+
return value.decode(errors="ignore")
|
|
44
|
+
return str(value)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _channel_names(channels: dict[int, dict[str, object]], *, field: str) -> list[str]:
|
|
48
|
+
names: list[str] = []
|
|
49
|
+
for key in sorted(channels):
|
|
50
|
+
meta = channels[key]
|
|
51
|
+
name = meta.get(field)
|
|
52
|
+
if name is None:
|
|
53
|
+
raise ParseError(
|
|
54
|
+
f"Channel metadata missing field '{field}' for channel {key}. Use channel_name_field: pns or pnn."
|
|
55
|
+
)
|
|
56
|
+
names.append(str(name))
|
|
57
|
+
return names
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def parse_fcs_file(
|
|
61
|
+
path: Path,
|
|
62
|
+
*,
|
|
63
|
+
channel_name_field: str,
|
|
64
|
+
channel_map: Mapping[str, str] | None = None,
|
|
65
|
+
drop_channels: set[str] | None = None,
|
|
66
|
+
sample_id_from: Literal["stem", "name"] = "stem",
|
|
67
|
+
time_value: float = 0.0,
|
|
68
|
+
) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
69
|
+
try:
|
|
70
|
+
from flowio import FlowData # noqa: PLC0415
|
|
71
|
+
except Exception as exc: # pragma: no cover - environment-specific
|
|
72
|
+
raise ParseError("flowio is required for ingest/flow_cytometer. Re-sync the core reader environment.") from exc
|
|
73
|
+
|
|
74
|
+
field = str(channel_name_field).lower().strip()
|
|
75
|
+
mapped_names = {str(k): str(v) for k, v in (channel_map or {}).items()}
|
|
76
|
+
dropped = {str(channel) for channel in (drop_channels or set())}
|
|
77
|
+
|
|
78
|
+
flow = FlowData(str(path))
|
|
79
|
+
event_count = int(flow.event_count)
|
|
80
|
+
channel_count = int(flow.channel_count)
|
|
81
|
+
raw_events = np.asarray(flow.events, dtype=float)
|
|
82
|
+
if raw_events.size != event_count * channel_count:
|
|
83
|
+
raise ParseError(
|
|
84
|
+
f"Unexpected event buffer size for {path.name}: "
|
|
85
|
+
f"{raw_events.size} values for {event_count} events × {channel_count} channels."
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
values = raw_events.reshape(event_count, channel_count)
|
|
89
|
+
channel_names = _channel_names(flow.channels, field=field)
|
|
90
|
+
if len(channel_names) != channel_count:
|
|
91
|
+
raise ParseError(
|
|
92
|
+
f"Channel count mismatch: metadata has {len(channel_names)} names, events have {channel_count}."
|
|
93
|
+
)
|
|
94
|
+
if mapped_names:
|
|
95
|
+
remapped = [mapped_names.get(name, name) for name in channel_names]
|
|
96
|
+
if len(set(remapped)) != len(remapped):
|
|
97
|
+
raise ParseError("channel_map produces duplicate channel names; ensure a 1:1 mapping.")
|
|
98
|
+
channel_names = remapped
|
|
99
|
+
|
|
100
|
+
wide = pd.DataFrame(values, columns=channel_names)
|
|
101
|
+
wide["event_index"] = range(event_count)
|
|
102
|
+
long = wide.melt(id_vars=["event_index"], var_name="channel", value_name="value")
|
|
103
|
+
if dropped:
|
|
104
|
+
long = long[~long["channel"].isin(dropped)]
|
|
105
|
+
sample_id = path.stem if sample_id_from == "stem" else path.name
|
|
106
|
+
long["sample_id"] = sample_id
|
|
107
|
+
long["position"] = sample_id
|
|
108
|
+
long["time"] = float(time_value)
|
|
109
|
+
|
|
110
|
+
channel_rows = []
|
|
111
|
+
channel_indices = sorted(flow.channels)
|
|
112
|
+
for idx, name in zip(channel_indices, channel_names, strict=False):
|
|
113
|
+
meta = flow.channels.get(idx, {})
|
|
114
|
+
pne_decades, pne_zero = _parse_pne(meta.get("pne"))
|
|
115
|
+
channel_rows.append(
|
|
116
|
+
{
|
|
117
|
+
"sample_id": sample_id,
|
|
118
|
+
"channel_index": int(idx),
|
|
119
|
+
"channel_name": str(name),
|
|
120
|
+
"pns": _clean_text(meta.get("pns")),
|
|
121
|
+
"pnn": _clean_text(meta.get("pnn")),
|
|
122
|
+
"pnt": _clean_text(meta.get("pnt")),
|
|
123
|
+
"pnf": _clean_text(meta.get("pnf")),
|
|
124
|
+
"pnl": _clean_text(meta.get("pnl")),
|
|
125
|
+
"pnr": _to_float(meta.get("pnr")),
|
|
126
|
+
"pnb": _to_float(meta.get("pnb")),
|
|
127
|
+
"png": _to_float(meta.get("png")),
|
|
128
|
+
"pne_decades": pne_decades,
|
|
129
|
+
"pne_zero": pne_zero,
|
|
130
|
+
}
|
|
131
|
+
)
|
|
132
|
+
channels_meta = pd.DataFrame(channel_rows)
|
|
133
|
+
if channels_meta.empty:
|
|
134
|
+
channels_meta = pd.DataFrame(columns=["sample_id", "channel_index", "channel_name"])
|
|
135
|
+
return long, channels_meta
|