reader-workbench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- reader_workbench/__init__.py +22 -0
- reader_workbench/__main__.py +4 -0
- reader_workbench/_version.py +17 -0
- reader_workbench/api/__init__.py +74 -0
- reader_workbench/api/_record_reads.py +75 -0
- reader_workbench/api/artifacts.py +79 -0
- reader_workbench/api/facade.py +538 -0
- reader_workbench/api/models.py +285 -0
- reader_workbench/api/notebooks.py +63 -0
- reader_workbench/contracts/__init__.py +18 -0
- reader_workbench/contracts/builtins/__init__.py +36 -0
- reader_workbench/contracts/builtins/cytometry.py +140 -0
- reader_workbench/contracts/builtins/four_state_event_window.py +214 -0
- reader_workbench/contracts/builtins/generic.py +21 -0
- reader_workbench/contracts/builtins/logic.py +149 -0
- reader_workbench/contracts/builtins/plate_reader.py +47 -0
- reader_workbench/contracts/catalog.py +257 -0
- reader_workbench/contracts/model.py +109 -0
- reader_workbench/domains/__init__.py +1 -0
- reader_workbench/domains/cytometry/__init__.py +3 -0
- reader_workbench/domains/cytometry/analysis/__init__.py +29 -0
- reader_workbench/domains/cytometry/analysis/events.py +182 -0
- reader_workbench/domains/cytometry/analysis/gating.py +175 -0
- reader_workbench/domains/cytometry/analysis/workflow.py +248 -0
- reader_workbench/domains/cytometry/io/__init__.py +3 -0
- reader_workbench/domains/cytometry/io/fcs.py +135 -0
- reader_workbench/domains/cytometry/plots/__init__.py +5 -0
- reader_workbench/domains/cytometry/plots/diagnostic.py +155 -0
- reader_workbench/domains/logic/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/pairs.py +661 -0
- reader_workbench/domains/logic/four_state_vector/__init__.py +7 -0
- reader_workbench/domains/logic/four_state_vector/builder.py +321 -0
- reader_workbench/domains/logic/four_state_vector/collection/__init__.py +20 -0
- reader_workbench/domains/logic/four_state_vector/collection/checks.py +49 -0
- reader_workbench/domains/logic/four_state_vector/collection/constants.py +29 -0
- reader_workbench/domains/logic/four_state_vector/collection/model.py +19 -0
- reader_workbench/domains/logic/four_state_vector/collection/render.py +383 -0
- reader_workbench/domains/logic/four_state_vector/collection/sources.py +185 -0
- reader_workbench/domains/logic/four_state_vector/config.py +214 -0
- reader_workbench/domains/logic/four_state_vector/diagnostic.py +361 -0
- reader_workbench/domains/logic/four_state_vector/heatmap.py +86 -0
- reader_workbench/domains/logic/four_state_vector/math.py +191 -0
- reader_workbench/domains/logic/four_state_vector/reference.py +85 -0
- reader_workbench/domains/logic/four_state_vector/selection.py +228 -0
- reader_workbench/domains/logic/four_state_vector/treatment_semantics.py +51 -0
- reader_workbench/domains/logic/four_state_vector/validation.py +19 -0
- reader_workbench/domains/logic/logic_symmetry/__init__.py +3 -0
- reader_workbench/domains/logic/logic_symmetry/encodings.py +93 -0
- reader_workbench/domains/logic/logic_symmetry/extract_corners.py +192 -0
- reader_workbench/domains/logic/logic_symmetry/main.py +236 -0
- reader_workbench/domains/logic/logic_symmetry/metrics.py +96 -0
- reader_workbench/domains/logic/logic_symmetry/overlay.py +129 -0
- reader_workbench/domains/logic/logic_symmetry/prep.py +138 -0
- reader_workbench/domains/logic/logic_symmetry/render.py +356 -0
- reader_workbench/domains/logic/treatment_columns.py +42 -0
- reader_workbench/domains/plate_reader/__init__.py +1 -0
- reader_workbench/domains/plate_reader/analysis/__init__.py +14 -0
- reader_workbench/domains/plate_reader/analysis/fold_change.py +474 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/__init__.py +21 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/aggregation.py +191 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contract_fields.py +50 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contracts.py +320 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/design_dispositions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/disposition_records.py +140 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/event_sensitivity.py +27 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/materialize.py +274 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/observation_resampling.py +97 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/reduction.py +62 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/seeds.py +15 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/sources.py +308 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusion_validation.py +94 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/timepoints.py +76 -0
- reader_workbench/domains/plate_reader/io/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/sample_map.py +65 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_kinetic.py +141 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_parser.py +295 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_shared.py +195 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_snapshot.py +130 -0
- reader_workbench/domains/plate_reader/ordering.py +59 -0
- reader_workbench/domains/plate_reader/plots/__init__.py +15 -0
- reader_workbench/domains/plate_reader/plots/_data.py +29 -0
- reader_workbench/domains/plate_reader/plots/common.py +346 -0
- reader_workbench/domains/plate_reader/plots/distributions.py +324 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych.py +525 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych_render.py +195 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/__init__.py +26 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic.py +295 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_components.py +149 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_render.py +308 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_style.py +51 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/schema.py +8 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/summary.py +140 -0
- reader_workbench/domains/plate_reader/plots/grouping.py +53 -0
- reader_workbench/domains/plate_reader/plots/panels/__init__.py +12 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot.py +161 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot_data.py +91 -0
- reader_workbench/domains/plate_reader/plots/panels/time_series.py +296 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic.py +435 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic_render.py +300 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/__init__.py +315 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/planning.py +168 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/__init__.py +205 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/inputs.py +103 -0
- reader_workbench/domains/plate_reader/plots/time_series.py +317 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/__init__.py +479 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/planning.py +283 -0
- reader_workbench/domains/time_series/__init__.py +29 -0
- reader_workbench/domains/time_series/aggregation.py +60 -0
- reader_workbench/domains/time_series/contracts.py +368 -0
- reader_workbench/domains/time_series/reduction.py +395 -0
- reader_workbench/errors.py +55 -0
- reader_workbench/maintenance/__init__.py +6 -0
- reader_workbench/maintenance/docs.py +335 -0
- reader_workbench/maintenance/model.py +28 -0
- reader_workbench/maintenance/release.py +39 -0
- reader_workbench/maintenance/skills.py +124 -0
- reader_workbench/plotting/__init__.py +20 -0
- reader_workbench/plotting/mpl.py +56 -0
- reader_workbench/plotting/sinks.py +69 -0
- reader_workbench/plotting/style.py +175 -0
- reader_workbench/plotting/utils.py +27 -0
- reader_workbench/plugins/__init__.py +1 -0
- reader_workbench/plugins/catalog.py +33 -0
- reader_workbench/plugins/export/__init__.py +0 -0
- reader_workbench/plugins/export/_paths.py +21 -0
- reader_workbench/plugins/export/csv.py +41 -0
- reader_workbench/plugins/export/xlsx.py +44 -0
- reader_workbench/plugins/ingest/__init__.py +0 -0
- reader_workbench/plugins/ingest/_discovery.py +58 -0
- reader_workbench/plugins/ingest/discovery_policy.py +66 -0
- reader_workbench/plugins/ingest/flow_cytometer.py +139 -0
- reader_workbench/plugins/ingest/synergy_h1.py +234 -0
- reader_workbench/plugins/manifests/__init__.py +1 -0
- reader_workbench/plugins/manifests/export.py +29 -0
- reader_workbench/plugins/manifests/ingest.py +29 -0
- reader_workbench/plugins/manifests/plot.py +161 -0
- reader_workbench/plugins/manifests/transform.py +172 -0
- reader_workbench/plugins/manifests/validator.py +18 -0
- reader_workbench/plugins/plot/__init__.py +0 -0
- reader_workbench/plugins/plot/_shared.py +55 -0
- reader_workbench/plugins/plot/cytometry_diagnostic.py +54 -0
- reader_workbench/plugins/plot/distributions.py +58 -0
- reader_workbench/plugins/plot/dual_reporter_triptych.py +224 -0
- reader_workbench/plugins/plot/four_state_event_window_diagnostic.py +133 -0
- reader_workbench/plugins/plot/four_state_event_window_summary.py +53 -0
- reader_workbench/plugins/plot/four_state_vector_collection.py +45 -0
- reader_workbench/plugins/plot/four_state_vector_diagnostic.py +105 -0
- reader_workbench/plugins/plot/four_state_vector_heatmap.py +63 -0
- reader_workbench/plugins/plot/logic_symmetry.py +56 -0
- reader_workbench/plugins/plot/single_reporter_diagnostic.py +279 -0
- reader_workbench/plugins/plot/snapshot_barplot.py +65 -0
- reader_workbench/plugins/plot/snapshot_heatmap.py +104 -0
- reader_workbench/plugins/plot/time_series.py +114 -0
- reader_workbench/plugins/plot/ts_and_snap.py +210 -0
- reader_workbench/plugins/transform/__init__.py +0 -0
- reader_workbench/plugins/transform/_four_state_vector.py +204 -0
- reader_workbench/plugins/transform/_labeling.py +109 -0
- reader_workbench/plugins/transform/alias.py +70 -0
- reader_workbench/plugins/transform/assay_labels.py +62 -0
- reader_workbench/plugins/transform/blank.py +79 -0
- reader_workbench/plugins/transform/crosstalk_pairs.py +180 -0
- reader_workbench/plugins/transform/cytometry_gating.py +120 -0
- reader_workbench/plugins/transform/fold_change.py +79 -0
- reader_workbench/plugins/transform/four_state_event_window.py +93 -0
- reader_workbench/plugins/transform/four_state_vector.py +62 -0
- reader_workbench/plugins/transform/four_state_vector_collection.py +41 -0
- reader_workbench/plugins/transform/logic_symmetry.py +67 -0
- reader_workbench/plugins/transform/outlier_filter.py +60 -0
- reader_workbench/plugins/transform/overflow.py +197 -0
- reader_workbench/plugins/transform/ratio.py +237 -0
- reader_workbench/plugins/transform/sample_map.py +170 -0
- reader_workbench/plugins/transform/sample_metadata.py +94 -0
- reader_workbench/plugins/validator/__init__.py +1 -0
- reader_workbench/plugins/validator/to_tidy_plus_map.py +155 -0
- reader_workbench/protocols/__init__.py +80 -0
- reader_workbench/protocols/_builtins_plate_reader_growth.py +179 -0
- reader_workbench/protocols/_builtins_plate_reader_variants.py +274 -0
- reader_workbench/protocols/builtins.py +1656 -0
- reader_workbench/protocols/compiler.py +22 -0
- reader_workbench/protocols/compilers/__init__.py +1 -0
- reader_workbench/protocols/compilers/common.py +100 -0
- reader_workbench/protocols/compilers/cytometry.py +87 -0
- reader_workbench/protocols/compilers/generic.py +14 -0
- reader_workbench/protocols/compilers/logic.py +245 -0
- reader_workbench/protocols/compilers/plate_reader.py +937 -0
- reader_workbench/protocols/compilers/plate_reader_pipeline.py +197 -0
- reader_workbench/protocols/model.py +1486 -0
- reader_workbench/protocols/semantic_coverage.py +234 -0
- reader_workbench/runtime/__init__.py +12 -0
- reader_workbench/runtime/builtin.py +23 -0
- reader_workbench/runtime/model.py +42 -0
- reader_workbench/workbench/__init__.py +60 -0
- reader_workbench/workbench/assets/__init__.py +22 -0
- reader_workbench/workbench/assets/types.py +118 -0
- reader_workbench/workbench/audit/__init__.py +5 -0
- reader_workbench/workbench/audit/experiments.py +307 -0
- reader_workbench/workbench/audit/staging.py +187 -0
- reader_workbench/workbench/cli/__init__.py +51 -0
- reader_workbench/workbench/cli/_lazy.py +9 -0
- reader_workbench/workbench/cli/_records_view.py +150 -0
- reader_workbench/workbench/cli/_surface_execution.py +443 -0
- reader_workbench/workbench/cli/audit.py +95 -0
- reader_workbench/workbench/cli/automation.py +229 -0
- reader_workbench/workbench/cli/demo.py +46 -0
- reader_workbench/workbench/cli/dop.py +91 -0
- reader_workbench/workbench/cli/experiments.py +635 -0
- reader_workbench/workbench/cli/helpers.py +232 -0
- reader_workbench/workbench/cli/main.py +59 -0
- reader_workbench/workbench/cli/maintenance.py +82 -0
- reader_workbench/workbench/cli/notebooks.py +260 -0
- reader_workbench/workbench/cli/pagination.py +117 -0
- reader_workbench/workbench/cli/protocols.py +336 -0
- reader_workbench/workbench/cli/shared.py +309 -0
- reader_workbench/workbench/cli/surfaces.py +534 -0
- reader_workbench/workbench/cli/verification.py +128 -0
- reader_workbench/workbench/commands.py +10 -0
- reader_workbench/workbench/config/__init__.py +47 -0
- reader_workbench/workbench/config/identity.py +13 -0
- reader_workbench/workbench/config/load.py +405 -0
- reader_workbench/workbench/config/model.py +274 -0
- reader_workbench/workbench/context.py +26 -0
- reader_workbench/workbench/decl/__init__.py +31 -0
- reader_workbench/workbench/decl/build.py +190 -0
- reader_workbench/workbench/decl/model.py +81 -0
- reader_workbench/workbench/dop/__init__.py +12 -0
- reader_workbench/workbench/dop/builtins.py +261 -0
- reader_workbench/workbench/dop/model.py +209 -0
- reader_workbench/workbench/engine/__init__.py +42 -0
- reader_workbench/workbench/engine/_shared.py +76 -0
- reader_workbench/workbench/engine/contracts.py +283 -0
- reader_workbench/workbench/engine/execution.py +326 -0
- reader_workbench/workbench/engine/file_outputs.py +260 -0
- reader_workbench/workbench/engine/inputs.py +161 -0
- reader_workbench/workbench/engine/invocations.py +507 -0
- reader_workbench/workbench/engine/planning.py +72 -0
- reader_workbench/workbench/engine/runtime.py +464 -0
- reader_workbench/workbench/engine/setup.py +149 -0
- reader_workbench/workbench/engine/validation.py +684 -0
- reader_workbench/workbench/experiment/__init__.py +47 -0
- reader_workbench/workbench/experiment/model.py +381 -0
- reader_workbench/workbench/experiments.py +133 -0
- reader_workbench/workbench/graph/__init__.py +47 -0
- reader_workbench/workbench/graph/nodes.py +102 -0
- reader_workbench/workbench/graph/normalize.py +177 -0
- reader_workbench/workbench/graph/refs.py +148 -0
- reader_workbench/workbench/input_discovery.py +19 -0
- reader_workbench/workbench/inspection/__init__.py +3 -0
- reader_workbench/workbench/inspection/catalogs.py +128 -0
- reader_workbench/workbench/inspection/common.py +92 -0
- reader_workbench/workbench/inspection/dop.py +64 -0
- reader_workbench/workbench/inspection/experiments.py +449 -0
- reader_workbench/workbench/inspection/inventory.py +68 -0
- reader_workbench/workbench/inspection/protocols.py +368 -0
- reader_workbench/workbench/inspection/readiness.py +333 -0
- reader_workbench/workbench/inspection/reports.py +367 -0
- reader_workbench/workbench/inspection/results.py +166 -0
- reader_workbench/workbench/inspection/runtime.py +287 -0
- reader_workbench/workbench/inspection/semantics.py +192 -0
- reader_workbench/workbench/inspection/validation.py +30 -0
- reader_workbench/workbench/notebooks/__init__.py +17 -0
- reader_workbench/workbench/notebooks/_launch_registry.py +112 -0
- reader_workbench/workbench/notebooks/_launch_runtime.py +104 -0
- reader_workbench/workbench/notebooks/components/__init__.py +21 -0
- reader_workbench/workbench/notebooks/components/deliverables.py +403 -0
- reader_workbench/workbench/notebooks/components/overview.py +119 -0
- reader_workbench/workbench/notebooks/eda.marimo.py.txt +153 -0
- reader_workbench/workbench/notebooks/launch.py +274 -0
- reader_workbench/workbench/notebooks/presentation.py +136 -0
- reader_workbench/workbench/notebooks/scaffold.py +60 -0
- reader_workbench/workbench/ontology.py +78 -0
- reader_workbench/workbench/paths.py +44 -0
- reader_workbench/workbench/ports/__init__.py +31 -0
- reader_workbench/workbench/ports/model.py +168 -0
- reader_workbench/workbench/records/__init__.py +44 -0
- reader_workbench/workbench/records/epoch.py +329 -0
- reader_workbench/workbench/records/evidence.py +247 -0
- reader_workbench/workbench/records/identity.py +87 -0
- reader_workbench/workbench/records/locking.py +185 -0
- reader_workbench/workbench/records/model.py +711 -0
- reader_workbench/workbench/records/sources.py +73 -0
- reader_workbench/workbench/records/store.py +1022 -0
- reader_workbench/workbench/records/verification.py +998 -0
- reader_workbench/workbench/registry.py +333 -0
- reader_workbench/workbench/spec_overrides.py +215 -0
- reader_workbench-1.0.0.dist-info/METADATA +91 -0
- reader_workbench-1.0.0.dist-info/RECORD +293 -0
- reader_workbench-1.0.0.dist-info/WHEEL +5 -0
- reader_workbench-1.0.0.dist-info/entry_points.txt +2 -0
- reader_workbench-1.0.0.dist-info/licenses/LICENSE +21 -0
- reader_workbench-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,383 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import math
|
|
4
|
+
import re
|
|
5
|
+
import textwrap
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
from reader_workbench.errors import FourStateVectorError
|
|
11
|
+
from reader_workbench.plotting.style import use_style
|
|
12
|
+
|
|
13
|
+
from .checks import require_normalized_frame
|
|
14
|
+
from .constants import VECTOR_CHANNELS
|
|
15
|
+
|
|
16
|
+
CHANNEL_LABELS = ("v00", "v10", "v01", "v11", "y00*", "y10*", "y01*", "y11*")
|
|
17
|
+
LOGIC_CHANNEL_COUNT = 4
|
|
18
|
+
LOGIC_COLORBAR_LABEL = "$v_i$ normalized response"
|
|
19
|
+
INTENSITY_COLORBAR_LABEL = "$y_i^\\star$ anchored log2 intensity"
|
|
20
|
+
NATURAL_SORT_TOKEN = re.compile(r"\d+|\D+")
|
|
21
|
+
TILE_SIZE_IN = 0.38
|
|
22
|
+
AXES_LEFT_MAX = 0.52
|
|
23
|
+
AXES_RIGHT = 0.94
|
|
24
|
+
AXES_BOTTOM = 0.24
|
|
25
|
+
AXES_TOP = 0.84
|
|
26
|
+
MIN_TICK_FONT_SIZE = 6.8
|
|
27
|
+
MAX_ROW_TICK_FONT_SIZE = 10.5
|
|
28
|
+
MAX_CHANNEL_TICK_FONT_SIZE = 12.5
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def render_four_state_vector_collection_heatmap(
|
|
32
|
+
frame: pd.DataFrame,
|
|
33
|
+
*,
|
|
34
|
+
title: str | None = None,
|
|
35
|
+
max_y_tick_labels: int = 80,
|
|
36
|
+
):
|
|
37
|
+
require_normalized_frame(frame)
|
|
38
|
+
try:
|
|
39
|
+
import matplotlib.pyplot as plt # noqa: PLC0415
|
|
40
|
+
from matplotlib.colors import LinearSegmentedColormap, Normalize, TwoSlopeNorm # noqa: PLC0415
|
|
41
|
+
except Exception as exc: # pragma: no cover - dependency guard
|
|
42
|
+
raise FourStateVectorError("four-state vector collection heatmap requires matplotlib.") from exc
|
|
43
|
+
|
|
44
|
+
plot_frame = _ordered_plot_frame(frame)
|
|
45
|
+
values = plot_frame.loc[:, list(VECTOR_CHANNELS)].astype(float)
|
|
46
|
+
row_labels = _display_row_labels(plot_frame)
|
|
47
|
+
row_count = len(values)
|
|
48
|
+
matrix = values.to_numpy()
|
|
49
|
+
figsize = _figure_size(row_labels, row_count=row_count)
|
|
50
|
+
row_tick_font_size = _row_tick_font_size(row_count=row_count, figure_height=figsize[1])
|
|
51
|
+
channel_tick_font_size = _channel_tick_font_size(figure_width=figsize[0])
|
|
52
|
+
visible_y_tick_labels = _visible_y_tick_label_count(
|
|
53
|
+
max_labels=max_y_tick_labels,
|
|
54
|
+
row_count=row_count,
|
|
55
|
+
figure_height=figsize[1],
|
|
56
|
+
font_size=row_tick_font_size,
|
|
57
|
+
)
|
|
58
|
+
annotation_font_size = _annotation_font_size(figure_width=figsize[0])
|
|
59
|
+
colorbar_font_size = _colorbar_font_size(row_tick_font_size)
|
|
60
|
+
with use_style(
|
|
61
|
+
{
|
|
62
|
+
"figure_figsize": figsize,
|
|
63
|
+
"axes_grid": False,
|
|
64
|
+
"xtick_labelsize": channel_tick_font_size,
|
|
65
|
+
"ytick_labelsize": row_tick_font_size,
|
|
66
|
+
"axes_titlesize": annotation_font_size,
|
|
67
|
+
"font_size": 12.0,
|
|
68
|
+
}
|
|
69
|
+
):
|
|
70
|
+
fig, ax = plt.subplots(figsize=figsize, constrained_layout=False)
|
|
71
|
+
fig.subplots_adjust(
|
|
72
|
+
left=_left_margin(row_labels, figure_width=figsize[0]),
|
|
73
|
+
right=AXES_RIGHT,
|
|
74
|
+
bottom=AXES_BOTTOM,
|
|
75
|
+
top=AXES_TOP,
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
x_edges = np.arange(len(VECTOR_CHANNELS) + 1)
|
|
79
|
+
y_edges = np.arange(row_count + 1)
|
|
80
|
+
logic_mesh = ax.pcolormesh(
|
|
81
|
+
x_edges,
|
|
82
|
+
y_edges,
|
|
83
|
+
_masked_channel_block(matrix, start=0, stop=LOGIC_CHANNEL_COUNT),
|
|
84
|
+
cmap=_logic_colormap(LinearSegmentedColormap),
|
|
85
|
+
norm=_logic_norm(matrix[:, :LOGIC_CHANNEL_COUNT], Normalize),
|
|
86
|
+
edgecolors="white",
|
|
87
|
+
linewidth=0.65,
|
|
88
|
+
shading="flat",
|
|
89
|
+
)
|
|
90
|
+
intensity_mesh = ax.pcolormesh(
|
|
91
|
+
x_edges,
|
|
92
|
+
y_edges,
|
|
93
|
+
_masked_channel_block(matrix, start=LOGIC_CHANNEL_COUNT, stop=len(VECTOR_CHANNELS)),
|
|
94
|
+
cmap=_intensity_colormap(LinearSegmentedColormap),
|
|
95
|
+
norm=_centered_norm(matrix[:, LOGIC_CHANNEL_COUNT:], TwoSlopeNorm),
|
|
96
|
+
edgecolors="white",
|
|
97
|
+
linewidth=0.65,
|
|
98
|
+
shading="flat",
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
ax.set_xlim(0, len(VECTOR_CHANNELS))
|
|
102
|
+
ax.set_ylim(row_count, 0)
|
|
103
|
+
ax.set_aspect("equal", adjustable="box")
|
|
104
|
+
ax.set_anchor("NW")
|
|
105
|
+
ax.set_ylabel("source :: design")
|
|
106
|
+
ax.set_xticks(np.arange(len(VECTOR_CHANNELS)) + 0.5)
|
|
107
|
+
ax.set_xticklabels(CHANNEL_LABELS, rotation=90, ha="center", va="top", fontsize=channel_tick_font_size)
|
|
108
|
+
_set_y_ticks(ax, row_labels, max_labels=visible_y_tick_labels, fontsize=row_tick_font_size)
|
|
109
|
+
_draw_channel_annotations(ax, font_size=annotation_font_size)
|
|
110
|
+
_draw_row_group_boundaries(ax, plot_frame["design_id"].astype(str).tolist())
|
|
111
|
+
_draw_heatmap_centered_title(fig, ax, _wrapped_title(title or "four-state vector collection"))
|
|
112
|
+
_draw_split_colorbars(fig, ax, logic_mesh, intensity_mesh, font_size=colorbar_font_size)
|
|
113
|
+
ax.tick_params(axis="both", length=0)
|
|
114
|
+
ax.axvline(LOGIC_CHANNEL_COUNT, color="white", linewidth=1.5)
|
|
115
|
+
for spine in ax.spines.values():
|
|
116
|
+
spine.set_visible(False)
|
|
117
|
+
return fig
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _figure_size(labels: list[str], *, row_count: int) -> tuple[float, float]:
|
|
121
|
+
label_width = _label_width(labels)
|
|
122
|
+
heatmap_width = len(VECTOR_CHANNELS) * TILE_SIZE_IN
|
|
123
|
+
side_width = 1.55
|
|
124
|
+
width = max(7.2, min(11.4, label_width + heatmap_width + side_width))
|
|
125
|
+
height = max(5.4, min(20.0, row_count * TILE_SIZE_IN + 3.05))
|
|
126
|
+
return (width, height)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _label_width(labels: list[str]) -> float:
|
|
130
|
+
max_label_len = max((len(label) for label in labels), default=0)
|
|
131
|
+
return max(3.0, min(4.6, 2.55 + 0.045 * max_label_len))
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _left_margin(labels: list[str], *, figure_width: float) -> float:
|
|
135
|
+
return min(AXES_LEFT_MAX, _label_width(labels) / figure_width)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _axis_height_inches(figure_height: float) -> float:
|
|
139
|
+
return max(1.0, float(figure_height) * (AXES_TOP - AXES_BOTTOM))
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _row_tick_font_size(*, row_count: int, figure_height: float) -> float:
|
|
143
|
+
row_pitch_points = _axis_height_inches(figure_height) * 72.0 / max(int(row_count), 1)
|
|
144
|
+
return max(MIN_TICK_FONT_SIZE, min(MAX_ROW_TICK_FONT_SIZE, row_pitch_points * 0.78))
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _channel_tick_font_size(*, figure_width: float) -> float:
|
|
148
|
+
channel_pitch_points = (
|
|
149
|
+
float(figure_width) * (AXES_RIGHT - _left_margin([], figure_width=figure_width)) * 72.0
|
|
150
|
+
) / len(VECTOR_CHANNELS)
|
|
151
|
+
return max(10.5, min(MAX_CHANNEL_TICK_FONT_SIZE, channel_pitch_points * 0.34))
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _visible_y_tick_label_count(*, max_labels: int, row_count: int, figure_height: float, font_size: float) -> int:
|
|
155
|
+
axis_points = _axis_height_inches(figure_height) * 72.0
|
|
156
|
+
capacity = max(1, math.floor(axis_points / (float(font_size) * 1.22)))
|
|
157
|
+
return max(1, min(int(max_labels), int(row_count), capacity))
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _annotation_font_size(*, figure_width: float) -> float:
|
|
161
|
+
return max(12.0, min(14.0, float(figure_width) * 1.10))
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _colorbar_font_size(row_tick_font_size: float) -> float:
|
|
165
|
+
return max(8.5, min(10.2, float(row_tick_font_size)))
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _wrapped_title(title: str) -> str:
|
|
169
|
+
return textwrap.fill(str(title), width=72)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _set_y_ticks(ax, labels: list[str], *, max_labels: int, fontsize: float) -> None:
|
|
173
|
+
n_rows = len(labels)
|
|
174
|
+
if n_rows <= max_labels:
|
|
175
|
+
ticks = list(range(n_rows))
|
|
176
|
+
else:
|
|
177
|
+
step = max(1, math.ceil(n_rows / max_labels))
|
|
178
|
+
ticks = list(range(0, n_rows, step))
|
|
179
|
+
if ticks[-1] != n_rows - 1:
|
|
180
|
+
ticks.append(n_rows - 1)
|
|
181
|
+
ax.set_yticks([tick + 0.5 for tick in ticks])
|
|
182
|
+
ax.set_yticklabels([labels[index] for index in ticks], fontsize=fontsize)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _draw_row_group_boundaries(ax, design_ids: list[str]) -> None:
|
|
186
|
+
previous = _row_group_key(design_ids[0])
|
|
187
|
+
for index, design_id in enumerate(design_ids[1:], start=1):
|
|
188
|
+
group = _row_group_key(design_id)
|
|
189
|
+
if group != previous:
|
|
190
|
+
ax.axhline(index, color="#ffffff", linewidth=2.0)
|
|
191
|
+
ax.axhline(index, color="#4f4f4f", linewidth=0.55, alpha=0.45)
|
|
192
|
+
previous = group
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _masked_channel_block(matrix: np.ndarray, *, start: int, stop: int) -> np.ma.MaskedArray:
|
|
196
|
+
mask = np.ones(matrix.shape, dtype=bool)
|
|
197
|
+
mask[:, start:stop] = False
|
|
198
|
+
return np.ma.masked_array(matrix, mask=mask)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _logic_norm(values: np.ndarray, normalize_type):
|
|
202
|
+
finite = np.asarray(values, dtype=float)
|
|
203
|
+
finite = finite[np.isfinite(finite)]
|
|
204
|
+
upper = max(1.0, float(finite.max())) if finite.size else 1.0
|
|
205
|
+
lower = min(0.0, float(finite.min())) if finite.size else 0.0
|
|
206
|
+
return normalize_type(vmin=lower, vmax=upper)
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _centered_norm(values: np.ndarray, norm_type):
|
|
210
|
+
finite = np.asarray(values, dtype=float)
|
|
211
|
+
finite = finite[np.isfinite(finite)]
|
|
212
|
+
limit = max(1.0, float(np.abs(finite).max())) if finite.size else 1.0
|
|
213
|
+
return norm_type(vmin=-limit, vcenter=0.0, vmax=limit)
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _logic_colormap(colormap_type):
|
|
217
|
+
cmap = colormap_type.from_list("four_state_vector_logic_blue", ("#f7f9fb", "#9bbfda", "#10306d"))
|
|
218
|
+
cmap.set_bad(color=(1.0, 1.0, 1.0, 0.0))
|
|
219
|
+
return cmap
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _intensity_colormap(colormap_type):
|
|
223
|
+
cmap = colormap_type.from_list("four_state_vector_intensity_diverging", ("#10306d", "#f7f7f2", "#c96845"))
|
|
224
|
+
cmap.set_bad(color=(1.0, 1.0, 1.0, 0.0))
|
|
225
|
+
return cmap
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _draw_channel_annotations(ax, *, font_size: float) -> None:
|
|
229
|
+
transform = ax.get_xaxis_transform()
|
|
230
|
+
ax.text(
|
|
231
|
+
LOGIC_CHANNEL_COUNT / 2,
|
|
232
|
+
1.13,
|
|
233
|
+
"Logic\npattern",
|
|
234
|
+
ha="center",
|
|
235
|
+
va="bottom",
|
|
236
|
+
linespacing=0.9,
|
|
237
|
+
fontsize=font_size,
|
|
238
|
+
transform=transform,
|
|
239
|
+
clip_on=False,
|
|
240
|
+
)
|
|
241
|
+
ax.text(
|
|
242
|
+
LOGIC_CHANNEL_COUNT + (len(VECTOR_CHANNELS) - LOGIC_CHANNEL_COUNT) / 2,
|
|
243
|
+
1.13,
|
|
244
|
+
"Anchored\nintensity",
|
|
245
|
+
ha="center",
|
|
246
|
+
va="bottom",
|
|
247
|
+
linespacing=0.9,
|
|
248
|
+
fontsize=font_size,
|
|
249
|
+
transform=transform,
|
|
250
|
+
clip_on=False,
|
|
251
|
+
)
|
|
252
|
+
for start, stop in ((0, LOGIC_CHANNEL_COUNT), (LOGIC_CHANNEL_COUNT, len(VECTOR_CHANNELS))):
|
|
253
|
+
ax.plot(
|
|
254
|
+
[start + 0.08, stop - 0.08],
|
|
255
|
+
[1.08, 1.08],
|
|
256
|
+
color="#8a8a8a",
|
|
257
|
+
linewidth=1.6,
|
|
258
|
+
transform=transform,
|
|
259
|
+
clip_on=False,
|
|
260
|
+
)
|
|
261
|
+
ax.text(
|
|
262
|
+
len(VECTOR_CHANNELS) / 2,
|
|
263
|
+
1.04,
|
|
264
|
+
r"vector = concat($v$, $y^\star$)",
|
|
265
|
+
ha="center",
|
|
266
|
+
va="top",
|
|
267
|
+
fontsize=max(11.0, font_size - 3.0),
|
|
268
|
+
transform=transform,
|
|
269
|
+
clip_on=False,
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _draw_heatmap_centered_title(fig, ax, title: str) -> None:
|
|
274
|
+
fig.canvas.draw()
|
|
275
|
+
box = ax.get_position()
|
|
276
|
+
fig.text(
|
|
277
|
+
(box.x0 + box.x1) / 2,
|
|
278
|
+
0.965,
|
|
279
|
+
title,
|
|
280
|
+
ha="center",
|
|
281
|
+
va="top",
|
|
282
|
+
fontsize=15.0,
|
|
283
|
+
)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _draw_split_colorbars(fig, ax, logic_mesh, intensity_mesh, *, font_size: float) -> None:
|
|
287
|
+
fig.canvas.draw()
|
|
288
|
+
box = ax.get_position()
|
|
289
|
+
colorbar_height = 0.014
|
|
290
|
+
# Anchor colorbars in the reserved bottom margin. Sparse heatmaps can shift
|
|
291
|
+
# the aspect-adjusted axes upward, which should not move the legend stack.
|
|
292
|
+
logic_y = AXES_BOTTOM - 0.070
|
|
293
|
+
intensity_y = max(0.035, logic_y - 0.055)
|
|
294
|
+
logic_cax = fig.add_axes([box.x0, logic_y, box.width, colorbar_height])
|
|
295
|
+
intensity_cax = fig.add_axes([box.x0, intensity_y, box.width, colorbar_height])
|
|
296
|
+
logic_cbar = fig.colorbar(logic_mesh, cax=logic_cax, orientation="horizontal")
|
|
297
|
+
logic_cbar.ax.xaxis.set_label_position("top")
|
|
298
|
+
logic_cbar.ax.xaxis.set_ticks_position("top")
|
|
299
|
+
logic_cbar.set_label(LOGIC_COLORBAR_LABEL, labelpad=4)
|
|
300
|
+
logic_cbar.set_ticks([0.0, 1.0])
|
|
301
|
+
intensity_cbar = fig.colorbar(intensity_mesh, cax=intensity_cax, orientation="horizontal")
|
|
302
|
+
intensity_cbar.set_label(INTENSITY_COLORBAR_LABEL, labelpad=4)
|
|
303
|
+
for colorbar in (logic_cbar, intensity_cbar):
|
|
304
|
+
colorbar.ax.tick_params(axis="x", labelsize=font_size, length=2)
|
|
305
|
+
colorbar.ax.xaxis.label.set_size(font_size)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def _ordered_plot_frame(frame: pd.DataFrame) -> pd.DataFrame:
|
|
309
|
+
order = sorted(range(len(frame)), key=lambda index: _row_sort_key(frame.iloc[index], fallback_index=index))
|
|
310
|
+
return frame.iloc[order].reset_index(drop=True)
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def _row_sort_key(row: pd.Series, *, fallback_index: int) -> tuple[object, ...]:
|
|
314
|
+
design_id = str(row["design_id"])
|
|
315
|
+
source_label = str(row[_source_label_column(row.index)])
|
|
316
|
+
return (_natural_sort_key(design_id), _natural_sort_key(source_label), fallback_index)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _row_group_key(design_id: str) -> tuple[object, ...]:
|
|
320
|
+
return _natural_sort_key(_design_family_label(design_id))
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _display_row_labels(frame: pd.DataFrame) -> list[str]:
|
|
324
|
+
source_labels = frame[_source_label_column(frame.columns)].astype(str).map(_short_source_label)
|
|
325
|
+
design_labels = frame["design_id"].astype(str).map(_short_design_label)
|
|
326
|
+
time_labels = frame["time_selected_h"].map(_timepoint_label) if "time_selected_h" in frame.columns else None
|
|
327
|
+
labels = []
|
|
328
|
+
for index, source_label in enumerate(source_labels.tolist()):
|
|
329
|
+
time_label = "" if time_labels is None else f" {time_labels.iloc[index]}"
|
|
330
|
+
labels.append(f"{source_label}{time_label} :: {design_labels.iloc[index]}")
|
|
331
|
+
if len(set(labels)) == len(labels):
|
|
332
|
+
return labels
|
|
333
|
+
fallback = frame["row_label"].astype(str)
|
|
334
|
+
if time_labels is not None:
|
|
335
|
+
fallback = fallback + " @ " + time_labels.astype(str)
|
|
336
|
+
return fallback.tolist()
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def _source_label_column(columns) -> str:
|
|
340
|
+
if "source_resource_id" in columns:
|
|
341
|
+
return "source_resource_id"
|
|
342
|
+
if "source_experiment_id" in columns:
|
|
343
|
+
return "source_experiment_id"
|
|
344
|
+
raise FourStateVectorError("four-state vector heatmap requires explicit source resource or experiment identity.")
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def _short_source_label(value: str) -> str:
|
|
348
|
+
return _middle_truncate(value.strip(), max_chars=24)
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _short_design_label(value: str) -> str:
|
|
352
|
+
return _middle_truncate(value.strip(), max_chars=28)
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
def _design_family_label(design_id: str) -> str:
|
|
356
|
+
text = design_id.strip()
|
|
357
|
+
match = re.match(r"[A-Za-z]+", text)
|
|
358
|
+
return match.group(0) if match else text
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def _natural_sort_key(value: str) -> tuple[tuple[int, object], ...]:
|
|
362
|
+
return tuple(
|
|
363
|
+
(0, int(token)) if token.isdigit() else (1, token.lower()) for token in NATURAL_SORT_TOKEN.findall(value)
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _timepoint_label(value: object) -> str:
|
|
368
|
+
try:
|
|
369
|
+
number = float(value)
|
|
370
|
+
except (TypeError, ValueError):
|
|
371
|
+
return f"t={value}"
|
|
372
|
+
if not math.isfinite(number):
|
|
373
|
+
return f"t={value}"
|
|
374
|
+
return f"t={number:.2f}h"
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def _middle_truncate(value: str, *, max_chars: int) -> str:
|
|
378
|
+
if len(value) <= max_chars:
|
|
379
|
+
return value
|
|
380
|
+
keep = max_chars - 3
|
|
381
|
+
head = keep // 2
|
|
382
|
+
tail = keep - head
|
|
383
|
+
return f"{value[:head]}...{value[-tail:]}"
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import math
|
|
4
|
+
from collections import Counter
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
import pandas as pd
|
|
8
|
+
|
|
9
|
+
from reader_workbench.errors import FourStateVectorError
|
|
10
|
+
|
|
11
|
+
from .checks import finite_numeric_column, require_vector_columns
|
|
12
|
+
from .constants import METADATA_COLUMNS, VECTOR_CHANNELS
|
|
13
|
+
from .model import FourStateVectorCollection, FourStateVectorSource
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def collect_four_state_vector_sources(
|
|
17
|
+
sources: list[FourStateVectorSource] | tuple[FourStateVectorSource, ...],
|
|
18
|
+
) -> FourStateVectorCollection:
|
|
19
|
+
"""Validate and combine already-resolved vector sources.
|
|
20
|
+
|
|
21
|
+
Resolving experiment configurations and record catalogs is a runtime concern.
|
|
22
|
+
This domain operation accepts explicit source data and owns only four-state vector
|
|
23
|
+
validation and normalization.
|
|
24
|
+
"""
|
|
25
|
+
if not sources:
|
|
26
|
+
raise FourStateVectorError("four-state vector collection requires at least one source.")
|
|
27
|
+
_require_unique_source_records(sources)
|
|
28
|
+
|
|
29
|
+
frames: list[pd.DataFrame] = []
|
|
30
|
+
for source_index, source in enumerate(sources):
|
|
31
|
+
normalized = _normalize_vector_frame(source, source_index=source_index)
|
|
32
|
+
frames.append(normalized)
|
|
33
|
+
|
|
34
|
+
frame = pd.concat(frames, ignore_index=True)
|
|
35
|
+
if frame.empty:
|
|
36
|
+
raise FourStateVectorError("four-state vector collection has no rows to plot.")
|
|
37
|
+
return FourStateVectorCollection(frame=frame)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _normalize_vector_frame(source: FourStateVectorSource, *, source_index: int) -> pd.DataFrame:
|
|
41
|
+
resource_id = _nonempty_identity(source.resource_id, field="resource_id")
|
|
42
|
+
experiment_id = _nonempty_identity(source.experiment_id, field="experiment_id")
|
|
43
|
+
record_id = _nonempty_identity(source.record_id, field="record_id")
|
|
44
|
+
revision_digest = _canonical_sha256_digest(source.revision_digest)
|
|
45
|
+
source_label = f"{resource_id} ({experiment_id}:{record_id})"
|
|
46
|
+
frame = source.frame.copy()
|
|
47
|
+
require_vector_columns(frame)
|
|
48
|
+
if frame.empty:
|
|
49
|
+
raise FourStateVectorError(f"four-state vector source has no rows: {source_label}")
|
|
50
|
+
|
|
51
|
+
out = frame.reset_index(drop=True)
|
|
52
|
+
for channel in VECTOR_CHANNELS:
|
|
53
|
+
out[channel] = finite_numeric_column(out[channel], column=channel, source=source_label)
|
|
54
|
+
out["design_id"] = _nonempty_string_column(out["design_id"], column="design_id", source=source_label)
|
|
55
|
+
if "time_selected_h" in out.columns:
|
|
56
|
+
out["time_selected_h"] = finite_numeric_column(
|
|
57
|
+
out["time_selected_h"],
|
|
58
|
+
column="time_selected_h",
|
|
59
|
+
source=source_label,
|
|
60
|
+
allow_nan=True,
|
|
61
|
+
)
|
|
62
|
+
out["reference_design_id"] = _nonempty_string_column(
|
|
63
|
+
out["reference_design_id"], column="reference_design_id", source=source_label
|
|
64
|
+
)
|
|
65
|
+
out["intensity_log2_offset_delta"] = _nonnegative_numeric_column(
|
|
66
|
+
out["intensity_log2_offset_delta"], column="intensity_log2_offset_delta", source=source_label
|
|
67
|
+
)
|
|
68
|
+
out["r_logic"] = _nonnegative_numeric_column(out["r_logic"], column="r_logic", source=source_label)
|
|
69
|
+
out["flat_logic"] = _strict_bool_column(out["flat_logic"], column="flat_logic", source=source_label)
|
|
70
|
+
if out["design_id"].duplicated().any():
|
|
71
|
+
duplicates = sorted(out.loc[out["design_id"].duplicated(keep=False), "design_id"].unique())
|
|
72
|
+
raise FourStateVectorError(
|
|
73
|
+
"four-state vector collection design_id values must be unique within each source: " + ", ".join(duplicates)
|
|
74
|
+
)
|
|
75
|
+
out.insert(0, "source_row_index", range(len(out)))
|
|
76
|
+
out.insert(0, "source_record_revision_digest", revision_digest)
|
|
77
|
+
out.insert(0, "source_record_id", record_id)
|
|
78
|
+
out.insert(0, "source_experiment_id", experiment_id)
|
|
79
|
+
out.insert(0, "source_resource_id", resource_id)
|
|
80
|
+
out.insert(0, "source_index", int(source_index))
|
|
81
|
+
out["row_label"] = _row_labels(out)
|
|
82
|
+
|
|
83
|
+
ordered = [
|
|
84
|
+
*METADATA_COLUMNS,
|
|
85
|
+
*[column for column in VECTOR_CHANNELS if column in out.columns],
|
|
86
|
+
]
|
|
87
|
+
ordered = [column for column in ordered if column in out.columns]
|
|
88
|
+
ordered += [column for column in out.columns if column not in set(ordered)]
|
|
89
|
+
return out.loc[:, ordered]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _require_unique_source_records(sources: list[FourStateVectorSource] | tuple[FourStateVectorSource, ...]) -> None:
|
|
93
|
+
identities = [
|
|
94
|
+
(
|
|
95
|
+
_nonempty_identity(source.experiment_id, field="experiment_id"),
|
|
96
|
+
_nonempty_identity(source.record_id, field="record_id"),
|
|
97
|
+
)
|
|
98
|
+
for source in sources
|
|
99
|
+
]
|
|
100
|
+
duplicates = sorted(identity for identity, count in Counter(identities).items() if count > 1)
|
|
101
|
+
if duplicates:
|
|
102
|
+
formatted = ", ".join(f"{experiment_id}:{record_id}" for experiment_id, record_id in duplicates)
|
|
103
|
+
raise FourStateVectorError(f"four-state vector collection source record identities must be unique: {formatted}")
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _nonempty_identity(value: str, *, field: str) -> str:
|
|
107
|
+
if not isinstance(value, str) or not value.strip():
|
|
108
|
+
raise FourStateVectorError(f"four-state vector collection {field} must be a non-empty string.")
|
|
109
|
+
return value.strip()
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _canonical_sha256_digest(value: str) -> str:
|
|
113
|
+
if isinstance(value, str) and value.startswith("sha256:"):
|
|
114
|
+
digest = value.removeprefix("sha256:")
|
|
115
|
+
if len(digest) == 64 and all(character in "0123456789abcdef" for character in digest):
|
|
116
|
+
return value
|
|
117
|
+
raise FourStateVectorError("four-state vector collection revision_digest must be a canonical sha256 digest.")
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _nonempty_string_column(series: pd.Series, *, column: str, source: str) -> pd.Series:
|
|
121
|
+
values = series.astype("string")
|
|
122
|
+
invalid = values.isna() | values.str.strip().eq("")
|
|
123
|
+
if invalid.any():
|
|
124
|
+
raise FourStateVectorError(
|
|
125
|
+
f"four-state vector collection column {column!r} must contain non-empty labels in {source}."
|
|
126
|
+
)
|
|
127
|
+
return values.astype(str)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _nonnegative_numeric_column(series: pd.Series, *, column: str, source: str) -> pd.Series:
|
|
131
|
+
values = finite_numeric_column(series, column=column, source=source)
|
|
132
|
+
if (values < 0.0).any():
|
|
133
|
+
raise FourStateVectorError(
|
|
134
|
+
f"four-state vector collection column {column!r} must contain nonnegative values in {source}."
|
|
135
|
+
)
|
|
136
|
+
return values
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _strict_bool_column(series: pd.Series, *, column: str, source: str) -> pd.Series:
|
|
140
|
+
parsed: list[bool] = []
|
|
141
|
+
invalid = False
|
|
142
|
+
for value in series.tolist():
|
|
143
|
+
if isinstance(value, bool):
|
|
144
|
+
parsed.append(value)
|
|
145
|
+
continue
|
|
146
|
+
if pd.isna(value):
|
|
147
|
+
invalid = True
|
|
148
|
+
break
|
|
149
|
+
if isinstance(value, str):
|
|
150
|
+
normalized = value.strip().lower()
|
|
151
|
+
if normalized == "true":
|
|
152
|
+
parsed.append(True)
|
|
153
|
+
continue
|
|
154
|
+
if normalized == "false":
|
|
155
|
+
parsed.append(False)
|
|
156
|
+
continue
|
|
157
|
+
invalid = True
|
|
158
|
+
break
|
|
159
|
+
if invalid:
|
|
160
|
+
raise FourStateVectorError(
|
|
161
|
+
f"four-state vector collection column {column!r} must contain boolean values in {source}."
|
|
162
|
+
)
|
|
163
|
+
return pd.Series(parsed, index=series.index, dtype=bool)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _row_labels(frame: pd.DataFrame) -> pd.Series:
|
|
167
|
+
labels = frame["source_resource_id"].astype(str) + " :: " + frame["design_id"].astype(str)
|
|
168
|
+
if not labels.duplicated().any():
|
|
169
|
+
return labels
|
|
170
|
+
if "time_selected_h" in frame.columns:
|
|
171
|
+
labels = labels + " @ " + frame["time_selected_h"].map(_format_time_label)
|
|
172
|
+
if not labels.duplicated().any():
|
|
173
|
+
return labels
|
|
174
|
+
duplicate_index = labels.groupby(labels).cumcount() + 1
|
|
175
|
+
return labels + " #" + duplicate_index.astype(str)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _format_time_label(value: Any) -> str:
|
|
179
|
+
try:
|
|
180
|
+
number = float(value)
|
|
181
|
+
except (TypeError, ValueError):
|
|
182
|
+
return str(value)
|
|
183
|
+
if not math.isfinite(number):
|
|
184
|
+
return str(value)
|
|
185
|
+
return f"{number:g}h"
|