reader-workbench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- reader_workbench/__init__.py +22 -0
- reader_workbench/__main__.py +4 -0
- reader_workbench/_version.py +17 -0
- reader_workbench/api/__init__.py +74 -0
- reader_workbench/api/_record_reads.py +75 -0
- reader_workbench/api/artifacts.py +79 -0
- reader_workbench/api/facade.py +538 -0
- reader_workbench/api/models.py +285 -0
- reader_workbench/api/notebooks.py +63 -0
- reader_workbench/contracts/__init__.py +18 -0
- reader_workbench/contracts/builtins/__init__.py +36 -0
- reader_workbench/contracts/builtins/cytometry.py +140 -0
- reader_workbench/contracts/builtins/four_state_event_window.py +214 -0
- reader_workbench/contracts/builtins/generic.py +21 -0
- reader_workbench/contracts/builtins/logic.py +149 -0
- reader_workbench/contracts/builtins/plate_reader.py +47 -0
- reader_workbench/contracts/catalog.py +257 -0
- reader_workbench/contracts/model.py +109 -0
- reader_workbench/domains/__init__.py +1 -0
- reader_workbench/domains/cytometry/__init__.py +3 -0
- reader_workbench/domains/cytometry/analysis/__init__.py +29 -0
- reader_workbench/domains/cytometry/analysis/events.py +182 -0
- reader_workbench/domains/cytometry/analysis/gating.py +175 -0
- reader_workbench/domains/cytometry/analysis/workflow.py +248 -0
- reader_workbench/domains/cytometry/io/__init__.py +3 -0
- reader_workbench/domains/cytometry/io/fcs.py +135 -0
- reader_workbench/domains/cytometry/plots/__init__.py +5 -0
- reader_workbench/domains/cytometry/plots/diagnostic.py +155 -0
- reader_workbench/domains/logic/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/pairs.py +661 -0
- reader_workbench/domains/logic/four_state_vector/__init__.py +7 -0
- reader_workbench/domains/logic/four_state_vector/builder.py +321 -0
- reader_workbench/domains/logic/four_state_vector/collection/__init__.py +20 -0
- reader_workbench/domains/logic/four_state_vector/collection/checks.py +49 -0
- reader_workbench/domains/logic/four_state_vector/collection/constants.py +29 -0
- reader_workbench/domains/logic/four_state_vector/collection/model.py +19 -0
- reader_workbench/domains/logic/four_state_vector/collection/render.py +383 -0
- reader_workbench/domains/logic/four_state_vector/collection/sources.py +185 -0
- reader_workbench/domains/logic/four_state_vector/config.py +214 -0
- reader_workbench/domains/logic/four_state_vector/diagnostic.py +361 -0
- reader_workbench/domains/logic/four_state_vector/heatmap.py +86 -0
- reader_workbench/domains/logic/four_state_vector/math.py +191 -0
- reader_workbench/domains/logic/four_state_vector/reference.py +85 -0
- reader_workbench/domains/logic/four_state_vector/selection.py +228 -0
- reader_workbench/domains/logic/four_state_vector/treatment_semantics.py +51 -0
- reader_workbench/domains/logic/four_state_vector/validation.py +19 -0
- reader_workbench/domains/logic/logic_symmetry/__init__.py +3 -0
- reader_workbench/domains/logic/logic_symmetry/encodings.py +93 -0
- reader_workbench/domains/logic/logic_symmetry/extract_corners.py +192 -0
- reader_workbench/domains/logic/logic_symmetry/main.py +236 -0
- reader_workbench/domains/logic/logic_symmetry/metrics.py +96 -0
- reader_workbench/domains/logic/logic_symmetry/overlay.py +129 -0
- reader_workbench/domains/logic/logic_symmetry/prep.py +138 -0
- reader_workbench/domains/logic/logic_symmetry/render.py +356 -0
- reader_workbench/domains/logic/treatment_columns.py +42 -0
- reader_workbench/domains/plate_reader/__init__.py +1 -0
- reader_workbench/domains/plate_reader/analysis/__init__.py +14 -0
- reader_workbench/domains/plate_reader/analysis/fold_change.py +474 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/__init__.py +21 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/aggregation.py +191 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contract_fields.py +50 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contracts.py +320 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/design_dispositions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/disposition_records.py +140 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/event_sensitivity.py +27 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/materialize.py +274 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/observation_resampling.py +97 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/reduction.py +62 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/seeds.py +15 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/sources.py +308 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusion_validation.py +94 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/timepoints.py +76 -0
- reader_workbench/domains/plate_reader/io/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/sample_map.py +65 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_kinetic.py +141 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_parser.py +295 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_shared.py +195 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_snapshot.py +130 -0
- reader_workbench/domains/plate_reader/ordering.py +59 -0
- reader_workbench/domains/plate_reader/plots/__init__.py +15 -0
- reader_workbench/domains/plate_reader/plots/_data.py +29 -0
- reader_workbench/domains/plate_reader/plots/common.py +346 -0
- reader_workbench/domains/plate_reader/plots/distributions.py +324 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych.py +525 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych_render.py +195 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/__init__.py +26 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic.py +295 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_components.py +149 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_render.py +308 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_style.py +51 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/schema.py +8 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/summary.py +140 -0
- reader_workbench/domains/plate_reader/plots/grouping.py +53 -0
- reader_workbench/domains/plate_reader/plots/panels/__init__.py +12 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot.py +161 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot_data.py +91 -0
- reader_workbench/domains/plate_reader/plots/panels/time_series.py +296 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic.py +435 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic_render.py +300 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/__init__.py +315 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/planning.py +168 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/__init__.py +205 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/inputs.py +103 -0
- reader_workbench/domains/plate_reader/plots/time_series.py +317 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/__init__.py +479 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/planning.py +283 -0
- reader_workbench/domains/time_series/__init__.py +29 -0
- reader_workbench/domains/time_series/aggregation.py +60 -0
- reader_workbench/domains/time_series/contracts.py +368 -0
- reader_workbench/domains/time_series/reduction.py +395 -0
- reader_workbench/errors.py +55 -0
- reader_workbench/maintenance/__init__.py +6 -0
- reader_workbench/maintenance/docs.py +335 -0
- reader_workbench/maintenance/model.py +28 -0
- reader_workbench/maintenance/release.py +39 -0
- reader_workbench/maintenance/skills.py +124 -0
- reader_workbench/plotting/__init__.py +20 -0
- reader_workbench/plotting/mpl.py +56 -0
- reader_workbench/plotting/sinks.py +69 -0
- reader_workbench/plotting/style.py +175 -0
- reader_workbench/plotting/utils.py +27 -0
- reader_workbench/plugins/__init__.py +1 -0
- reader_workbench/plugins/catalog.py +33 -0
- reader_workbench/plugins/export/__init__.py +0 -0
- reader_workbench/plugins/export/_paths.py +21 -0
- reader_workbench/plugins/export/csv.py +41 -0
- reader_workbench/plugins/export/xlsx.py +44 -0
- reader_workbench/plugins/ingest/__init__.py +0 -0
- reader_workbench/plugins/ingest/_discovery.py +58 -0
- reader_workbench/plugins/ingest/discovery_policy.py +66 -0
- reader_workbench/plugins/ingest/flow_cytometer.py +139 -0
- reader_workbench/plugins/ingest/synergy_h1.py +234 -0
- reader_workbench/plugins/manifests/__init__.py +1 -0
- reader_workbench/plugins/manifests/export.py +29 -0
- reader_workbench/plugins/manifests/ingest.py +29 -0
- reader_workbench/plugins/manifests/plot.py +161 -0
- reader_workbench/plugins/manifests/transform.py +172 -0
- reader_workbench/plugins/manifests/validator.py +18 -0
- reader_workbench/plugins/plot/__init__.py +0 -0
- reader_workbench/plugins/plot/_shared.py +55 -0
- reader_workbench/plugins/plot/cytometry_diagnostic.py +54 -0
- reader_workbench/plugins/plot/distributions.py +58 -0
- reader_workbench/plugins/plot/dual_reporter_triptych.py +224 -0
- reader_workbench/plugins/plot/four_state_event_window_diagnostic.py +133 -0
- reader_workbench/plugins/plot/four_state_event_window_summary.py +53 -0
- reader_workbench/plugins/plot/four_state_vector_collection.py +45 -0
- reader_workbench/plugins/plot/four_state_vector_diagnostic.py +105 -0
- reader_workbench/plugins/plot/four_state_vector_heatmap.py +63 -0
- reader_workbench/plugins/plot/logic_symmetry.py +56 -0
- reader_workbench/plugins/plot/single_reporter_diagnostic.py +279 -0
- reader_workbench/plugins/plot/snapshot_barplot.py +65 -0
- reader_workbench/plugins/plot/snapshot_heatmap.py +104 -0
- reader_workbench/plugins/plot/time_series.py +114 -0
- reader_workbench/plugins/plot/ts_and_snap.py +210 -0
- reader_workbench/plugins/transform/__init__.py +0 -0
- reader_workbench/plugins/transform/_four_state_vector.py +204 -0
- reader_workbench/plugins/transform/_labeling.py +109 -0
- reader_workbench/plugins/transform/alias.py +70 -0
- reader_workbench/plugins/transform/assay_labels.py +62 -0
- reader_workbench/plugins/transform/blank.py +79 -0
- reader_workbench/plugins/transform/crosstalk_pairs.py +180 -0
- reader_workbench/plugins/transform/cytometry_gating.py +120 -0
- reader_workbench/plugins/transform/fold_change.py +79 -0
- reader_workbench/plugins/transform/four_state_event_window.py +93 -0
- reader_workbench/plugins/transform/four_state_vector.py +62 -0
- reader_workbench/plugins/transform/four_state_vector_collection.py +41 -0
- reader_workbench/plugins/transform/logic_symmetry.py +67 -0
- reader_workbench/plugins/transform/outlier_filter.py +60 -0
- reader_workbench/plugins/transform/overflow.py +197 -0
- reader_workbench/plugins/transform/ratio.py +237 -0
- reader_workbench/plugins/transform/sample_map.py +170 -0
- reader_workbench/plugins/transform/sample_metadata.py +94 -0
- reader_workbench/plugins/validator/__init__.py +1 -0
- reader_workbench/plugins/validator/to_tidy_plus_map.py +155 -0
- reader_workbench/protocols/__init__.py +80 -0
- reader_workbench/protocols/_builtins_plate_reader_growth.py +179 -0
- reader_workbench/protocols/_builtins_plate_reader_variants.py +274 -0
- reader_workbench/protocols/builtins.py +1656 -0
- reader_workbench/protocols/compiler.py +22 -0
- reader_workbench/protocols/compilers/__init__.py +1 -0
- reader_workbench/protocols/compilers/common.py +100 -0
- reader_workbench/protocols/compilers/cytometry.py +87 -0
- reader_workbench/protocols/compilers/generic.py +14 -0
- reader_workbench/protocols/compilers/logic.py +245 -0
- reader_workbench/protocols/compilers/plate_reader.py +937 -0
- reader_workbench/protocols/compilers/plate_reader_pipeline.py +197 -0
- reader_workbench/protocols/model.py +1486 -0
- reader_workbench/protocols/semantic_coverage.py +234 -0
- reader_workbench/runtime/__init__.py +12 -0
- reader_workbench/runtime/builtin.py +23 -0
- reader_workbench/runtime/model.py +42 -0
- reader_workbench/workbench/__init__.py +60 -0
- reader_workbench/workbench/assets/__init__.py +22 -0
- reader_workbench/workbench/assets/types.py +118 -0
- reader_workbench/workbench/audit/__init__.py +5 -0
- reader_workbench/workbench/audit/experiments.py +307 -0
- reader_workbench/workbench/audit/staging.py +187 -0
- reader_workbench/workbench/cli/__init__.py +51 -0
- reader_workbench/workbench/cli/_lazy.py +9 -0
- reader_workbench/workbench/cli/_records_view.py +150 -0
- reader_workbench/workbench/cli/_surface_execution.py +443 -0
- reader_workbench/workbench/cli/audit.py +95 -0
- reader_workbench/workbench/cli/automation.py +229 -0
- reader_workbench/workbench/cli/demo.py +46 -0
- reader_workbench/workbench/cli/dop.py +91 -0
- reader_workbench/workbench/cli/experiments.py +635 -0
- reader_workbench/workbench/cli/helpers.py +232 -0
- reader_workbench/workbench/cli/main.py +59 -0
- reader_workbench/workbench/cli/maintenance.py +82 -0
- reader_workbench/workbench/cli/notebooks.py +260 -0
- reader_workbench/workbench/cli/pagination.py +117 -0
- reader_workbench/workbench/cli/protocols.py +336 -0
- reader_workbench/workbench/cli/shared.py +309 -0
- reader_workbench/workbench/cli/surfaces.py +534 -0
- reader_workbench/workbench/cli/verification.py +128 -0
- reader_workbench/workbench/commands.py +10 -0
- reader_workbench/workbench/config/__init__.py +47 -0
- reader_workbench/workbench/config/identity.py +13 -0
- reader_workbench/workbench/config/load.py +405 -0
- reader_workbench/workbench/config/model.py +274 -0
- reader_workbench/workbench/context.py +26 -0
- reader_workbench/workbench/decl/__init__.py +31 -0
- reader_workbench/workbench/decl/build.py +190 -0
- reader_workbench/workbench/decl/model.py +81 -0
- reader_workbench/workbench/dop/__init__.py +12 -0
- reader_workbench/workbench/dop/builtins.py +261 -0
- reader_workbench/workbench/dop/model.py +209 -0
- reader_workbench/workbench/engine/__init__.py +42 -0
- reader_workbench/workbench/engine/_shared.py +76 -0
- reader_workbench/workbench/engine/contracts.py +283 -0
- reader_workbench/workbench/engine/execution.py +326 -0
- reader_workbench/workbench/engine/file_outputs.py +260 -0
- reader_workbench/workbench/engine/inputs.py +161 -0
- reader_workbench/workbench/engine/invocations.py +507 -0
- reader_workbench/workbench/engine/planning.py +72 -0
- reader_workbench/workbench/engine/runtime.py +464 -0
- reader_workbench/workbench/engine/setup.py +149 -0
- reader_workbench/workbench/engine/validation.py +684 -0
- reader_workbench/workbench/experiment/__init__.py +47 -0
- reader_workbench/workbench/experiment/model.py +381 -0
- reader_workbench/workbench/experiments.py +133 -0
- reader_workbench/workbench/graph/__init__.py +47 -0
- reader_workbench/workbench/graph/nodes.py +102 -0
- reader_workbench/workbench/graph/normalize.py +177 -0
- reader_workbench/workbench/graph/refs.py +148 -0
- reader_workbench/workbench/input_discovery.py +19 -0
- reader_workbench/workbench/inspection/__init__.py +3 -0
- reader_workbench/workbench/inspection/catalogs.py +128 -0
- reader_workbench/workbench/inspection/common.py +92 -0
- reader_workbench/workbench/inspection/dop.py +64 -0
- reader_workbench/workbench/inspection/experiments.py +449 -0
- reader_workbench/workbench/inspection/inventory.py +68 -0
- reader_workbench/workbench/inspection/protocols.py +368 -0
- reader_workbench/workbench/inspection/readiness.py +333 -0
- reader_workbench/workbench/inspection/reports.py +367 -0
- reader_workbench/workbench/inspection/results.py +166 -0
- reader_workbench/workbench/inspection/runtime.py +287 -0
- reader_workbench/workbench/inspection/semantics.py +192 -0
- reader_workbench/workbench/inspection/validation.py +30 -0
- reader_workbench/workbench/notebooks/__init__.py +17 -0
- reader_workbench/workbench/notebooks/_launch_registry.py +112 -0
- reader_workbench/workbench/notebooks/_launch_runtime.py +104 -0
- reader_workbench/workbench/notebooks/components/__init__.py +21 -0
- reader_workbench/workbench/notebooks/components/deliverables.py +403 -0
- reader_workbench/workbench/notebooks/components/overview.py +119 -0
- reader_workbench/workbench/notebooks/eda.marimo.py.txt +153 -0
- reader_workbench/workbench/notebooks/launch.py +274 -0
- reader_workbench/workbench/notebooks/presentation.py +136 -0
- reader_workbench/workbench/notebooks/scaffold.py +60 -0
- reader_workbench/workbench/ontology.py +78 -0
- reader_workbench/workbench/paths.py +44 -0
- reader_workbench/workbench/ports/__init__.py +31 -0
- reader_workbench/workbench/ports/model.py +168 -0
- reader_workbench/workbench/records/__init__.py +44 -0
- reader_workbench/workbench/records/epoch.py +329 -0
- reader_workbench/workbench/records/evidence.py +247 -0
- reader_workbench/workbench/records/identity.py +87 -0
- reader_workbench/workbench/records/locking.py +185 -0
- reader_workbench/workbench/records/model.py +711 -0
- reader_workbench/workbench/records/sources.py +73 -0
- reader_workbench/workbench/records/store.py +1022 -0
- reader_workbench/workbench/records/verification.py +998 -0
- reader_workbench/workbench/registry.py +333 -0
- reader_workbench/workbench/spec_overrides.py +215 -0
- reader_workbench-1.0.0.dist-info/METADATA +91 -0
- reader_workbench-1.0.0.dist-info/RECORD +293 -0
- reader_workbench-1.0.0.dist-info/WHEEL +5 -0
- reader_workbench-1.0.0.dist-info/entry_points.txt +2 -0
- reader_workbench-1.0.0.dist-info/licenses/LICENSE +21 -0
- reader_workbench-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
"""four-state vector math: build vector from logic/intensity per-corner points.
|
|
2
|
+
|
|
3
|
+
Compute the vector = [v00,v10,v01,v11, y00*,y10*,y01*,y11*] per design.
|
|
4
|
+
|
|
5
|
+
- v* are computed from the LOGIC CHANNEL (e.g., YFP/CFP):
|
|
6
|
+
# r_logic and v computation are derived from the LOGIC channel corner means.
|
|
7
|
+
# r_logic is the dynamic range on the linear scale (after ε guard),
|
|
8
|
+
# while v is obtained by log2 + min-max on the *log* scale.
|
|
9
|
+
if max(u)-min(u) <= eps_range: v_i = 0.25
|
|
10
|
+
else: v = (u - u_min) / ((u_max - u_min) + eta)
|
|
11
|
+
|
|
12
|
+
- y* are computed from the INTENSITY CHANNEL (e.g., YFP/OD600):
|
|
13
|
+
y_linear_i = (b_i + eps_abs) / max(anchor_i + ref_add_alpha, eps_ref)
|
|
14
|
+
y*_i = log2( max(y_linear_i + log2_offset_delta, eps_ratio) )
|
|
15
|
+
|
|
16
|
+
Anchors are computed from the INTENSITY per-corner table for the reference design_id."""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
import pandas as pd
|
|
22
|
+
from pandas.api.types import is_bool_dtype
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _safe_log2(x: np.ndarray | float, eps: float) -> np.ndarray | float:
|
|
26
|
+
return np.log2(np.maximum(x, eps))
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _logic_minmax_from_four(
|
|
30
|
+
vals: tuple[float, float, float, float], *, eps_ratio: float, eps_range: float, eta: float
|
|
31
|
+
) -> tuple[np.ndarray, float, bool, float, float, float, str, str]:
|
|
32
|
+
a = np.array(vals, dtype=float)
|
|
33
|
+
# ε‑guarded linear values (used for r_logic = max/min)
|
|
34
|
+
a_guard = np.maximum(a, eps_ratio)
|
|
35
|
+
# log2 for min–max shape mapping
|
|
36
|
+
u = _safe_log2(a_guard, eps_ratio)
|
|
37
|
+
umin, umax = float(np.min(u)), float(np.max(u))
|
|
38
|
+
span = umax - umin
|
|
39
|
+
flat = bool(span <= eps_range)
|
|
40
|
+
if flat:
|
|
41
|
+
v = np.full(4, 0.25, dtype=float)
|
|
42
|
+
else:
|
|
43
|
+
denom = span + float(eta)
|
|
44
|
+
if not np.isfinite(denom) or denom <= 0.0:
|
|
45
|
+
raise ValueError(f"four-state vector: invalid logic min-max denom (span={span}, eta={eta}).")
|
|
46
|
+
v = np.clip((u - umin) / denom, 0.0, 1.0)
|
|
47
|
+
M = float(np.max(a_guard))
|
|
48
|
+
m = float(np.min(a_guard))
|
|
49
|
+
r = (M / m) if M > 0 and m > 0 else 1.0
|
|
50
|
+
# Corner labels for clarity in logs/diagnostics
|
|
51
|
+
corners = np.array(["00", "10", "01", "11"])
|
|
52
|
+
cmax = str(corners[int(np.argmax(a_guard))])
|
|
53
|
+
cmin = str(corners[int(np.argmin(a_guard))])
|
|
54
|
+
# Return v, r, flat, plus diagnostics:
|
|
55
|
+
# M (max), m (min) on linear scale; span on log2 scale; and which corners hit them.
|
|
56
|
+
return v.astype(float), r, flat, M, m, span, cmax, cmin
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def compute_four_state_vector(
|
|
60
|
+
*,
|
|
61
|
+
points_logic: pd.DataFrame, # b00..b11 from LOGIC channel
|
|
62
|
+
points_intensity: pd.DataFrame, # b00..b11 from INTENSITY channel
|
|
63
|
+
per_corner_intensity: pd.DataFrame, # per-corner table for anchors
|
|
64
|
+
design_by: list[str],
|
|
65
|
+
reference_design_id: str | None,
|
|
66
|
+
reference_stat: str,
|
|
67
|
+
eps_ratio: float,
|
|
68
|
+
eps_range: float,
|
|
69
|
+
eps_ref: float, # hard lower bound for (A + α)
|
|
70
|
+
eps_abs: float, # small add to numerator b_i (absolute intensity)
|
|
71
|
+
ref_add_alpha: float, # α
|
|
72
|
+
log2_offset_delta: float, # δ
|
|
73
|
+
) -> pd.DataFrame:
|
|
74
|
+
label_col = design_by[0]
|
|
75
|
+
|
|
76
|
+
# anchors from INTENSITY per-corner table for the reference design_id
|
|
77
|
+
ref_tab: pd.DataFrame | None = None
|
|
78
|
+
if reference_design_id:
|
|
79
|
+
ref_rows = per_corner_intensity[per_corner_intensity[label_col].astype(str) == str(reference_design_id)].copy()
|
|
80
|
+
if not ref_rows.empty:
|
|
81
|
+
agg_fun = "median" if reference_stat == "median" else "mean"
|
|
82
|
+
ref_tab = (
|
|
83
|
+
ref_rows.groupby(["corner"])["y_mean"]
|
|
84
|
+
.agg(agg_fun)
|
|
85
|
+
.reset_index()
|
|
86
|
+
.rename(columns={"y_mean": "anchor_mean"})
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
idx_cols = design_by
|
|
90
|
+
L = points_logic.set_index(idx_cols)
|
|
91
|
+
I_vec = points_intensity.set_index(idx_cols)
|
|
92
|
+
merged = (
|
|
93
|
+
L[["b00", "b10", "b01", "b11"]]
|
|
94
|
+
.join(I_vec[["b00", "b10", "b01", "b11"]], how="inner", lsuffix="_logic", rsuffix="_intensity")
|
|
95
|
+
.reset_index()
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
out_rows: list[dict[str, object]] = []
|
|
99
|
+
for _, row in merged.iterrows():
|
|
100
|
+
v, r_logic, flat, rmax, rmin, span_log2, cmax, cmin = _logic_minmax_from_four(
|
|
101
|
+
(float(row["b00_logic"]), float(row["b10_logic"]), float(row["b01_logic"]), float(row["b11_logic"])),
|
|
102
|
+
eps_ratio=eps_ratio,
|
|
103
|
+
eps_range=eps_range,
|
|
104
|
+
eta=eps_range,
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
anchors: dict[str, float] = {"00": np.nan, "10": np.nan, "01": np.nan, "11": np.nan}
|
|
108
|
+
if ref_tab is not None and not ref_tab.empty:
|
|
109
|
+
for _, rr in ref_tab.iterrows():
|
|
110
|
+
anchors[str(rr["corner"])] = float(rr["anchor_mean"])
|
|
111
|
+
|
|
112
|
+
def ystar(b: float, a: float, *, corner: str, design_label: object) -> float:
|
|
113
|
+
# Spec-aligned:
|
|
114
|
+
# denom = max(a + α, eps_ref)
|
|
115
|
+
# y_linear = (b + eps_abs) / denom
|
|
116
|
+
# y* = log2( max(y_linear + δ, eps_ratio) )
|
|
117
|
+
anchor = float(a)
|
|
118
|
+
sample = float(b)
|
|
119
|
+
if not np.isfinite(anchor):
|
|
120
|
+
raise ValueError(
|
|
121
|
+
f"four-state vector: missing anchor or non-finite anchor for corner {corner}; check reference configuration"
|
|
122
|
+
)
|
|
123
|
+
if not np.isfinite(sample):
|
|
124
|
+
raise ValueError(
|
|
125
|
+
f"four-state vector: non-finite intensity for design {design_label!r} corner {corner}; check input data"
|
|
126
|
+
)
|
|
127
|
+
denom = max(anchor + float(ref_add_alpha), float(eps_ref))
|
|
128
|
+
if not np.isfinite(denom) or denom <= 0:
|
|
129
|
+
raise ValueError(f"four-state vector: invalid anchor denominator (max(A+alpha, eps_ref))={denom}")
|
|
130
|
+
y_linear = (sample + float(eps_abs)) / denom
|
|
131
|
+
log_arg = y_linear + float(log2_offset_delta)
|
|
132
|
+
# Guard only for the log argument (assertive, no silent backfills elsewhere)
|
|
133
|
+
return float(np.log2(np.maximum(log_arg, float(eps_ratio))))
|
|
134
|
+
|
|
135
|
+
design_label = row[idx_cols[0]] if idx_cols else "<unknown>"
|
|
136
|
+
y00 = ystar(float(row["b00_intensity"]), anchors["00"], corner="00", design_label=design_label)
|
|
137
|
+
y10 = ystar(float(row["b10_intensity"]), anchors["10"], corner="10", design_label=design_label)
|
|
138
|
+
y01 = ystar(float(row["b01_intensity"]), anchors["01"], corner="01", design_label=design_label)
|
|
139
|
+
y11 = ystar(float(row["b11_intensity"]), anchors["11"], corner="11", design_label=design_label)
|
|
140
|
+
|
|
141
|
+
rec: dict[str, object] = {c: row[c] for c in idx_cols}
|
|
142
|
+
rec.update(
|
|
143
|
+
v00=float(v[0]),
|
|
144
|
+
v10=float(v[1]),
|
|
145
|
+
v01=float(v[2]),
|
|
146
|
+
v11=float(v[3]),
|
|
147
|
+
y00_star=y00,
|
|
148
|
+
y10_star=y10,
|
|
149
|
+
y01_star=y01,
|
|
150
|
+
y11_star=y11,
|
|
151
|
+
r_logic=r_logic,
|
|
152
|
+
flat_logic=flat,
|
|
153
|
+
# Self-describing diagnostics for r_logic:
|
|
154
|
+
r_logic_min=float(rmin),
|
|
155
|
+
r_logic_max=float(rmax),
|
|
156
|
+
logic_span_log2=float(span_log2),
|
|
157
|
+
r_logic_corner_min=cmin,
|
|
158
|
+
r_logic_corner_max=cmax,
|
|
159
|
+
)
|
|
160
|
+
out_rows.append(rec)
|
|
161
|
+
|
|
162
|
+
df = pd.DataFrame.from_records(out_rows)
|
|
163
|
+
|
|
164
|
+
# Keep numerics as float for stability and downstream math.
|
|
165
|
+
float_cols = [
|
|
166
|
+
"v00",
|
|
167
|
+
"v10",
|
|
168
|
+
"v01",
|
|
169
|
+
"v11",
|
|
170
|
+
"y00_star",
|
|
171
|
+
"y10_star",
|
|
172
|
+
"y01_star",
|
|
173
|
+
"y11_star",
|
|
174
|
+
"r_logic",
|
|
175
|
+
"r_logic_min",
|
|
176
|
+
"r_logic_max",
|
|
177
|
+
"logic_span_log2",
|
|
178
|
+
]
|
|
179
|
+
for c in float_cols:
|
|
180
|
+
if c in df.columns:
|
|
181
|
+
df[c] = pd.to_numeric(df[c], errors="coerce").astype(float)
|
|
182
|
+
|
|
183
|
+
# Contract compliance: flat_logic must be a proper boolean dtype.
|
|
184
|
+
if "flat_logic" in df.columns:
|
|
185
|
+
df["flat_logic"] = df["flat_logic"].astype(bool)
|
|
186
|
+
if not is_bool_dtype(df["flat_logic"]):
|
|
187
|
+
raise TypeError(
|
|
188
|
+
f"four-state vector internal error: expected boolean dtype for 'flat_logic', got {df['flat_logic'].dtype!r}"
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
return df
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Literal
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def _reduce(series: pd.Series, stat: Literal["mean", "median"]) -> float:
|
|
9
|
+
s = pd.to_numeric(series, errors="coerce").dropna()
|
|
10
|
+
if s.empty:
|
|
11
|
+
return float("nan")
|
|
12
|
+
return float(s.median()) if stat == "median" else float(s.mean())
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def resolve_reference_design_id(
|
|
16
|
+
df: pd.DataFrame,
|
|
17
|
+
*,
|
|
18
|
+
design_by: list[str],
|
|
19
|
+
ref_label: str | None,
|
|
20
|
+
) -> str | None:
|
|
21
|
+
"""
|
|
22
|
+
Deterministically resolve the configured reference label to the *raw* design label.
|
|
23
|
+
Policy:
|
|
24
|
+
1) If ref_label matches the raw label column exactly → use it.
|
|
25
|
+
2) Else, if <label>_alias exists and matches exactly → map to the single raw label.
|
|
26
|
+
3) Else → raise a clear error (no silent fallback).
|
|
27
|
+
"""
|
|
28
|
+
if ref_label is None:
|
|
29
|
+
return None
|
|
30
|
+
label_col = design_by[0] if design_by else "design_id"
|
|
31
|
+
alias_col = f"{label_col}_alias"
|
|
32
|
+
want = str(ref_label)
|
|
33
|
+
|
|
34
|
+
if label_col in df.columns and df[label_col].astype(str).eq(want).any():
|
|
35
|
+
return want # exact raw match
|
|
36
|
+
|
|
37
|
+
if alias_col in df.columns:
|
|
38
|
+
pairs = df[[label_col, alias_col]].dropna().astype(str).drop_duplicates()
|
|
39
|
+
matches = pairs.loc[pairs[alias_col] == want, label_col].unique()
|
|
40
|
+
if len(matches) == 1:
|
|
41
|
+
return str(matches[0]) # unique alias→raw mapping
|
|
42
|
+
if len(matches) > 1:
|
|
43
|
+
raise ValueError(
|
|
44
|
+
f"four_state_vector: reference.{label_col} {want!r} resolves via {alias_col!r} to multiple {label_col!r} values: "
|
|
45
|
+
f"{sorted(map(str, matches))!r}"
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
# diagnostics (short previews)
|
|
49
|
+
raw_vals = df[label_col].astype(str).unique().tolist() if label_col in df.columns else []
|
|
50
|
+
alias_vals = df[alias_col].astype(str).unique().tolist() if alias_col in df.columns else []
|
|
51
|
+
raw_prev = ", ".join(sorted(raw_vals)[:8]) + (" …" if len(raw_vals) > 8 else "")
|
|
52
|
+
alias_prev = ", ".join(sorted(alias_vals)[:8]) + (" …" if len(alias_vals) > 8 else "")
|
|
53
|
+
raise ValueError(
|
|
54
|
+
f"four_state_vector: reference.{label_col} {want!r} not found under {label_col!r} or {alias_col!r}.\n"
|
|
55
|
+
f" available raw: [{raw_prev}]\n"
|
|
56
|
+
f" available alias: [{alias_prev or '—'}]"
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def compute_reference_table(
|
|
61
|
+
per_corner: pd.DataFrame,
|
|
62
|
+
*,
|
|
63
|
+
design_by: list[str],
|
|
64
|
+
ref_design_id: str,
|
|
65
|
+
stat: Literal["mean", "median"] = "mean",
|
|
66
|
+
) -> pd.DataFrame:
|
|
67
|
+
"""
|
|
68
|
+
Build a table of reference b00/b10/b01/b11 values.
|
|
69
|
+
|
|
70
|
+
Returns a DataFrame:
|
|
71
|
+
one row with columns ['b00','b10','b01','b11']
|
|
72
|
+
"""
|
|
73
|
+
if not design_by:
|
|
74
|
+
raise ValueError("design_by must contain at least one column to locate the reference")
|
|
75
|
+
|
|
76
|
+
label_col = design_by[0]
|
|
77
|
+
ref_rows = per_corner[per_corner[label_col].astype(str) == str(ref_design_id)].copy()
|
|
78
|
+
if ref_rows.empty:
|
|
79
|
+
raise ValueError(f"Reference design_id {ref_design_id!r} has no rows in per-corner table.")
|
|
80
|
+
|
|
81
|
+
agg = ref_rows.groupby("corner")["y_mean"].agg(lambda s: _reduce(s, stat))
|
|
82
|
+
mean = pd.DataFrame([agg.to_dict()])
|
|
83
|
+
# Rename columns to b00..b11
|
|
84
|
+
mean = mean.rename(columns={"00": "b00", "10": "b10", "01": "b01", "11": "b11"})
|
|
85
|
+
return mean
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
"""four-state vector selection: time picking and corner-level observation aggregation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
from reader_workbench.domains.logic.treatment_columns import choose_treatment_column, normalize_treatment_series
|
|
11
|
+
|
|
12
|
+
REQUIRED_COLS = ["position", "time", "channel", "value"]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class CornerizeResult:
|
|
17
|
+
per_corner: pd.DataFrame # one row per (design×corner)
|
|
18
|
+
points: pd.DataFrame # wide one row per design with b00.., sd00.., n00..
|
|
19
|
+
chosen_time: float | None
|
|
20
|
+
time_warning: str | None = None
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _enforce_columns(df: pd.DataFrame, design_by: list[str]) -> None:
|
|
24
|
+
miss = [c for c in REQUIRED_COLS if c not in df.columns]
|
|
25
|
+
if miss:
|
|
26
|
+
raise ValueError(f"four-state vector: tidy data missing required columns: {miss}")
|
|
27
|
+
m2 = [c for c in design_by if c not in df.columns]
|
|
28
|
+
if m2:
|
|
29
|
+
raise ValueError(f"four-state vector: missing design_by columns: {m2}")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _pick_time(times: np.ndarray, target: float | None, mode: str) -> float | None:
|
|
33
|
+
if times.size == 0:
|
|
34
|
+
return None
|
|
35
|
+
times = np.unique(times.astype(float))
|
|
36
|
+
if target is None:
|
|
37
|
+
return float(np.max(times))
|
|
38
|
+
if mode == "exact":
|
|
39
|
+
hits = times[np.isclose(times, float(target), rtol=0, atol=1e-12)]
|
|
40
|
+
return float(hits[0]) if hits.size else None
|
|
41
|
+
if mode == "nearest":
|
|
42
|
+
return float(times[np.argmin(np.abs(times - float(target)))])
|
|
43
|
+
if mode == "last_before":
|
|
44
|
+
candidates = times[times <= float(target)]
|
|
45
|
+
return float(candidates.max()) if candidates.size else None
|
|
46
|
+
if mode == "first_after":
|
|
47
|
+
candidates = times[times >= float(target)]
|
|
48
|
+
return float(candidates.min()) if candidates.size else None
|
|
49
|
+
raise ValueError(f"Unknown time mode {mode!r}")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def select_times(
|
|
53
|
+
df: pd.DataFrame,
|
|
54
|
+
*,
|
|
55
|
+
channel: str,
|
|
56
|
+
treatment_map: dict[str, str],
|
|
57
|
+
case_sensitive: bool,
|
|
58
|
+
target_time_h: float | None,
|
|
59
|
+
time_mode: str,
|
|
60
|
+
tolerance_h: float | None,
|
|
61
|
+
preferred_treatment_column: str | None = None,
|
|
62
|
+
) -> tuple[pd.DataFrame, float | None, str | None]:
|
|
63
|
+
work = df[df["channel"] == channel].copy()
|
|
64
|
+
# Decide which column to match against (raw preferred; alias tolerated).
|
|
65
|
+
treatment_col = choose_treatment_column(
|
|
66
|
+
work,
|
|
67
|
+
treatment_map,
|
|
68
|
+
case_sensitive=case_sensitive,
|
|
69
|
+
preferred=preferred_treatment_column,
|
|
70
|
+
)
|
|
71
|
+
if case_sensitive:
|
|
72
|
+
mapped = [str(v) for v in treatment_map.values()]
|
|
73
|
+
work = work[work[treatment_col].astype(str).isin(mapped)].copy()
|
|
74
|
+
else:
|
|
75
|
+
mapped = [str(v).strip().casefold() for v in treatment_map.values()]
|
|
76
|
+
norm_col = "__norm_treatment"
|
|
77
|
+
work[norm_col] = normalize_treatment_series(work[treatment_col])
|
|
78
|
+
work = work[work[norm_col].isin(mapped)].copy()
|
|
79
|
+
work["_t_norm"] = work[norm_col]
|
|
80
|
+
|
|
81
|
+
if work.empty:
|
|
82
|
+
# Helpful, assertive diagnostics
|
|
83
|
+
present_unfiltered = df[df["channel"] == channel]
|
|
84
|
+
present_vals = (
|
|
85
|
+
present_unfiltered[treatment_col].astype(str).dropna().unique().tolist()
|
|
86
|
+
if treatment_col in present_unfiltered.columns
|
|
87
|
+
else []
|
|
88
|
+
)
|
|
89
|
+
preview_present = ", ".join(map(str, present_vals[:8])) + (" …" if len(present_vals) > 8 else "")
|
|
90
|
+
preview_expected = ", ".join(map(str, list(treatment_map.values())))
|
|
91
|
+
raise ValueError(
|
|
92
|
+
"four-state vector: no rows matched the requested channel/treatment_map.\n"
|
|
93
|
+
f" channel: {channel!r}\n"
|
|
94
|
+
f" using treatment column: {treatment_col!r}\n"
|
|
95
|
+
f" expected (from treatment_map): [{preview_expected}]\n"
|
|
96
|
+
f" present (in data at channel): [{preview_present}]\n"
|
|
97
|
+
"Hints: ensure aliases map raw treatments to the configured labels, or update treatment_map to "
|
|
98
|
+
"match the values in the chosen treatment column."
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
t = _pick_time(work["time"].to_numpy(dtype=float), target_time_h, time_mode)
|
|
102
|
+
if t is None:
|
|
103
|
+
raise ValueError("four-state vector: could not choose a global time.")
|
|
104
|
+
out = work[np.isclose(work["time"].astype(float), float(t), rtol=0, atol=1e-9)].copy()
|
|
105
|
+
|
|
106
|
+
warn_note = None
|
|
107
|
+
if tolerance_h is not None and target_time_h is not None:
|
|
108
|
+
tol = float(tolerance_h)
|
|
109
|
+
tgt = float(target_time_h)
|
|
110
|
+
delta = abs(float(t) - tgt)
|
|
111
|
+
if delta > tol:
|
|
112
|
+
warn_note = (
|
|
113
|
+
f"Config requested time={tgt:.3f} h (mode={time_mode}, tol={tol:.3f} h) "
|
|
114
|
+
f"but closest time={float(t):.3f} h (|Δ|={delta:.3f} h); using {float(t):.3f} h."
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
return out, float(t), warn_note
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def cornerize_and_aggregate(
|
|
121
|
+
df: pd.DataFrame,
|
|
122
|
+
*,
|
|
123
|
+
design_by: list[str],
|
|
124
|
+
treatment_map: dict[str, str],
|
|
125
|
+
case_sensitive: bool,
|
|
126
|
+
time_column: str,
|
|
127
|
+
channel: str,
|
|
128
|
+
target_time_h: float | None,
|
|
129
|
+
time_mode: str,
|
|
130
|
+
time_tolerance_h: float | None,
|
|
131
|
+
require_all_corners_per_design: bool,
|
|
132
|
+
preferred_treatment_column: str | None = None,
|
|
133
|
+
) -> CornerizeResult:
|
|
134
|
+
if time_column != "time":
|
|
135
|
+
if time_column not in df.columns:
|
|
136
|
+
raise ValueError(f"four-state vector: time column '{time_column}' not found.")
|
|
137
|
+
df = df.copy()
|
|
138
|
+
df["time"] = df[time_column]
|
|
139
|
+
|
|
140
|
+
_enforce_columns(df, design_by)
|
|
141
|
+
|
|
142
|
+
snap, chosen_time, warn_note = select_times(
|
|
143
|
+
df,
|
|
144
|
+
channel=channel,
|
|
145
|
+
treatment_map=treatment_map,
|
|
146
|
+
case_sensitive=case_sensitive,
|
|
147
|
+
preferred_treatment_column=preferred_treatment_column,
|
|
148
|
+
target_time_h=target_time_h,
|
|
149
|
+
time_mode=time_mode,
|
|
150
|
+
tolerance_h=time_tolerance_h,
|
|
151
|
+
)
|
|
152
|
+
if snap.empty:
|
|
153
|
+
raise ValueError("four-state vector: no rows remain after time selection.")
|
|
154
|
+
|
|
155
|
+
rev: dict[str, str] = {}
|
|
156
|
+
for k in ("00", "10", "01", "11"):
|
|
157
|
+
v = treatment_map[k]
|
|
158
|
+
key = str(v) if case_sensitive else str(v).strip().casefold()
|
|
159
|
+
if key in rev:
|
|
160
|
+
raise ValueError(f"four-state vector: duplicate treatment_map value: {v!r}")
|
|
161
|
+
rev[key] = k
|
|
162
|
+
|
|
163
|
+
# Map treatments → {00,10,01,11} using the same column we matched on.
|
|
164
|
+
corner_source = choose_treatment_column(
|
|
165
|
+
snap,
|
|
166
|
+
treatment_map,
|
|
167
|
+
case_sensitive=case_sensitive,
|
|
168
|
+
preferred=preferred_treatment_column,
|
|
169
|
+
)
|
|
170
|
+
if case_sensitive:
|
|
171
|
+
corner_values = snap[corner_source].astype(str)
|
|
172
|
+
rev_keys = {str(k): v for k, v in rev.items()}
|
|
173
|
+
else:
|
|
174
|
+
corner_values = normalize_treatment_series(snap[corner_source])
|
|
175
|
+
rev_keys = {str(k).strip().casefold(): v for k, v in rev.items()}
|
|
176
|
+
snap["corner"] = corner_values.map(rev_keys)
|
|
177
|
+
|
|
178
|
+
chk = snap.groupby(design_by + ["corner"])["time"].nunique().reset_index(name="n_times")
|
|
179
|
+
bad = chk[chk["n_times"] > 1]
|
|
180
|
+
if not bad.empty:
|
|
181
|
+
raise ValueError("four-state vector: more than one time within (design×corner) after selection.")
|
|
182
|
+
|
|
183
|
+
def _agg_mean(series: pd.Series) -> float:
|
|
184
|
+
s = pd.to_numeric(series, errors="coerce").dropna()
|
|
185
|
+
return float(s.mean()) if s.size else float("nan")
|
|
186
|
+
|
|
187
|
+
def _agg_sd(series: pd.Series) -> float:
|
|
188
|
+
s = pd.to_numeric(series, errors="coerce").dropna()
|
|
189
|
+
return float(s.std(ddof=1)) if s.size >= 2 else 0.0
|
|
190
|
+
|
|
191
|
+
def _agg_n(series: pd.Series) -> int:
|
|
192
|
+
return int(pd.to_numeric(series, errors="coerce").dropna().size)
|
|
193
|
+
|
|
194
|
+
grp_cols = design_by + ["corner"]
|
|
195
|
+
g = snap.groupby(grp_cols, dropna=False)
|
|
196
|
+
per_corner = g.agg(
|
|
197
|
+
time=("time", "first"), y_mean=("value", _agg_mean), y_sd=("value", _agg_sd), y_n=("value", _agg_n)
|
|
198
|
+
).reset_index()
|
|
199
|
+
|
|
200
|
+
idx_cols = design_by
|
|
201
|
+
m = per_corner.pivot_table(index=idx_cols, columns="corner", values="y_mean", aggfunc="first")
|
|
202
|
+
s = per_corner.pivot_table(index=idx_cols, columns="corner", values="y_sd", aggfunc="first")
|
|
203
|
+
n = per_corner.pivot_table(index=idx_cols, columns="corner", values="y_n", aggfunc="first")
|
|
204
|
+
|
|
205
|
+
req = ["00", "10", "01", "11"]
|
|
206
|
+
if require_all_corners_per_design:
|
|
207
|
+
missing_groups = []
|
|
208
|
+
for idx, row in m.iterrows():
|
|
209
|
+
miss = [c for c in req if pd.isna(row.get(c))]
|
|
210
|
+
if miss:
|
|
211
|
+
key = idx if isinstance(idx, tuple) else (idx,)
|
|
212
|
+
missing_groups.append(f"{dict(zip(idx_cols, key, strict=False))} → missing corners {miss}")
|
|
213
|
+
if missing_groups:
|
|
214
|
+
raise ValueError("four-state vector: incomplete corner sets:\n" + "\n".join(missing_groups[:40]))
|
|
215
|
+
|
|
216
|
+
points = (
|
|
217
|
+
m[req]
|
|
218
|
+
.rename(columns={"00": "b00", "10": "b10", "01": "b01", "11": "b11"})
|
|
219
|
+
.join(s[req].rename(columns={"00": "sd00", "10": "sd10", "01": "sd01", "11": "sd11"}))
|
|
220
|
+
.join(n[req].rename(columns={"00": "n00", "10": "n10", "01": "n01", "11": "n11"}))
|
|
221
|
+
.reset_index()
|
|
222
|
+
)
|
|
223
|
+
return CornerizeResult(
|
|
224
|
+
per_corner=per_corner,
|
|
225
|
+
points=points,
|
|
226
|
+
chosen_time=chosen_time,
|
|
227
|
+
time_warning=warn_note,
|
|
228
|
+
)
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Bind metric-neutral experiment states to the four-state four-state vector contract."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping, Sequence
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True)
|
|
11
|
+
class FourStateVectorTreatmentSemantics:
|
|
12
|
+
treatment_column: str
|
|
13
|
+
corners: dict[str, str]
|
|
14
|
+
case_sensitive: bool
|
|
15
|
+
|
|
16
|
+
def inject(self, config: dict[str, Any]) -> dict[str, Any]:
|
|
17
|
+
"""Return a copy of ``config`` with the resolved treatment contract."""
|
|
18
|
+
|
|
19
|
+
resolved = dict(config)
|
|
20
|
+
resolved["treatment_column"] = self.treatment_column
|
|
21
|
+
resolved["treatment_map"] = dict(self.corners)
|
|
22
|
+
resolved["treatment_case_sensitive"] = self.case_sensitive
|
|
23
|
+
return resolved
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def bind_four_state_vector_treatment_semantics(
|
|
27
|
+
*,
|
|
28
|
+
state_ids: Sequence[str],
|
|
29
|
+
source_column: str,
|
|
30
|
+
source_values: Mapping[str, str],
|
|
31
|
+
case_sensitive: bool,
|
|
32
|
+
treatment_column: str | None = None,
|
|
33
|
+
) -> FourStateVectorTreatmentSemantics:
|
|
34
|
+
"""Validate explicit state-space values for the four-state four-state vector transform."""
|
|
35
|
+
|
|
36
|
+
normalized_state_ids = tuple(str(state_id) for state_id in state_ids)
|
|
37
|
+
if normalized_state_ids != ("00", "10", "01", "11"):
|
|
38
|
+
raise ValueError("four-state vector state space must declare exactly 00, 10, 01, 11 in that order")
|
|
39
|
+
if set(source_values) != set(normalized_state_ids):
|
|
40
|
+
raise ValueError("four-state vector state values must define exactly 00, 10, 01, and 11")
|
|
41
|
+
column = treatment_column or source_column
|
|
42
|
+
if not isinstance(column, str) or not column.strip():
|
|
43
|
+
raise ValueError("four_state_vector treatment column must be a non-empty string")
|
|
44
|
+
return FourStateVectorTreatmentSemantics(
|
|
45
|
+
treatment_column=column.strip(),
|
|
46
|
+
corners={state_id: str(source_values[state_id]) for state_id in normalized_state_ids},
|
|
47
|
+
case_sensitive=bool(case_sensitive),
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
__all__ = ["FourStateVectorTreatmentSemantics", "bind_four_state_vector_treatment_semantics"]
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Shared validation for strict four-state vector dataframe contracts."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
from reader_workbench.errors import FourStateVectorError
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def require_intensity_delta_column(frame: pd.DataFrame) -> None:
|
|
11
|
+
column = "intensity_log2_offset_delta"
|
|
12
|
+
if column not in frame.columns:
|
|
13
|
+
raise FourStateVectorError(
|
|
14
|
+
f"four-state vector input requires column {column!r}. "
|
|
15
|
+
"Regenerate an logic.four_state_vector.v1 table instead of inferring a default."
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
__all__ = ["require_intensity_delta_column"]
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass(frozen=True)
|
|
10
|
+
class EncodingConfig:
|
|
11
|
+
size_by: str # "log_r" | "cv" | "fixed"
|
|
12
|
+
size_fixed: float
|
|
13
|
+
hue: str | None # None | column name
|
|
14
|
+
alpha_by: str | None
|
|
15
|
+
alpha_min: float
|
|
16
|
+
alpha_max: float
|
|
17
|
+
shape_by: str | None
|
|
18
|
+
shape_cycle: list[str]
|
|
19
|
+
shape_max_categories: int | None
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _scale_to_range(values: pd.Series, vmin: float, vmax: float, lo: float, hi: float) -> pd.Series:
|
|
23
|
+
if values.empty or np.all(values == values.iloc[0]):
|
|
24
|
+
return pd.Series([lo] * len(values), index=values.index, dtype=float)
|
|
25
|
+
# guard inf/nan
|
|
26
|
+
s = values.replace([np.inf, -np.inf], np.nan).fillna(vmin)
|
|
27
|
+
s = s.clip(lower=vmin, upper=vmax)
|
|
28
|
+
norm = (s - vmin) / max(vmax - vmin, 1e-12)
|
|
29
|
+
return lo + norm * (hi - lo)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _compute_size(df: pd.DataFrame, cfg: EncodingConfig) -> pd.Series:
|
|
33
|
+
if cfg.size_by == "fixed":
|
|
34
|
+
return pd.Series([float(cfg.size_fixed)] * len(df), index=df.index, dtype=float)
|
|
35
|
+
elif cfg.size_by == "log_r":
|
|
36
|
+
# Robustly scale log_r into visually distinct point areas
|
|
37
|
+
v = df["log_r"].clip(lower=0.0)
|
|
38
|
+
vmax = float(np.nanpercentile(v, 95)) if len(v) else 1.0
|
|
39
|
+
return _scale_to_range(v, 0.0, max(vmax, 1e-6), 40.0, 300.0)
|
|
40
|
+
elif cfg.size_by == "cv":
|
|
41
|
+
v = df["cv"].clip(lower=0.0)
|
|
42
|
+
vmax = float(np.nanpercentile(v, 95)) if len(v) else 1.0
|
|
43
|
+
return _scale_to_range(v, 0.0, max(vmax, 1e-6), 40.0, 300.0)
|
|
44
|
+
raise ValueError(f"Unknown size_by '{cfg.size_by}'")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _compute_alpha(df: pd.DataFrame, cfg: EncodingConfig) -> pd.Series:
|
|
48
|
+
if not cfg.alpha_by:
|
|
49
|
+
return pd.Series([cfg.alpha_max] * len(df), index=df.index, dtype=float)
|
|
50
|
+
col = cfg.alpha_by
|
|
51
|
+
if col not in df.columns:
|
|
52
|
+
raise ValueError(f"alpha_by refers to missing column '{col}'")
|
|
53
|
+
# Order categories numerically if possible (batch typically numeric)
|
|
54
|
+
series = df[col]
|
|
55
|
+
try:
|
|
56
|
+
order = np.sort(series.astype(float).unique())
|
|
57
|
+
order = [float(x) for x in order]
|
|
58
|
+
order_map = {v: i for i, v in enumerate(order)}
|
|
59
|
+
keys = series.astype(float).map(order_map)
|
|
60
|
+
except Exception:
|
|
61
|
+
cats = pd.Categorical(series.astype(str))
|
|
62
|
+
keys = pd.Series(cats.codes, index=series.index) # -1 for NaN
|
|
63
|
+
kmin, kmax = float(keys.min()), float(keys.max())
|
|
64
|
+
return _scale_to_range(keys.astype(float), kmin, kmax, cfg.alpha_min, cfg.alpha_max)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _compute_shape(df: pd.DataFrame, cfg: EncodingConfig) -> pd.Series:
|
|
68
|
+
if not cfg.shape_by:
|
|
69
|
+
return pd.Series(["o"] * len(df), index=df.index, dtype=object)
|
|
70
|
+
col = cfg.shape_by
|
|
71
|
+
if col not in df.columns:
|
|
72
|
+
raise ValueError(f"shape_by refers to missing column '{col}'")
|
|
73
|
+
cats = pd.Categorical(df[col].astype(str))
|
|
74
|
+
ncat = int(len(cats.categories))
|
|
75
|
+
if cfg.shape_max_categories is not None and ncat > int(cfg.shape_max_categories):
|
|
76
|
+
raise ValueError(
|
|
77
|
+
f"shape_by={col!r} has {ncat} categories, exceeding shape_max_categories={cfg.shape_max_categories}"
|
|
78
|
+
)
|
|
79
|
+
if ncat > len(cfg.shape_cycle):
|
|
80
|
+
raise ValueError(
|
|
81
|
+
f"Not enough markers in shape_cycle ({len(cfg.shape_cycle)}) for {ncat} categories; extend the cycle or lower categories."
|
|
82
|
+
)
|
|
83
|
+
mapping = {cat: cfg.shape_cycle[i] for i, cat in enumerate(cats.categories)}
|
|
84
|
+
return pd.Series([mapping[str(v)] for v in cats.astype(str)], index=df.index, dtype=object)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def apply_encodings(df_points: pd.DataFrame, cfg: EncodingConfig) -> pd.DataFrame:
|
|
88
|
+
out = df_points.copy()
|
|
89
|
+
out["size_value"] = _compute_size(out, cfg)
|
|
90
|
+
out["alpha_value"] = _compute_alpha(out, cfg)
|
|
91
|
+
out["hue_value"] = (out[cfg.hue] if cfg.hue else pd.Series([None] * len(out), index=out.index)).astype(object)
|
|
92
|
+
out["shape_value"] = _compute_shape(out, cfg)
|
|
93
|
+
return out
|