reader-workbench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- reader_workbench/__init__.py +22 -0
- reader_workbench/__main__.py +4 -0
- reader_workbench/_version.py +17 -0
- reader_workbench/api/__init__.py +74 -0
- reader_workbench/api/_record_reads.py +75 -0
- reader_workbench/api/artifacts.py +79 -0
- reader_workbench/api/facade.py +538 -0
- reader_workbench/api/models.py +285 -0
- reader_workbench/api/notebooks.py +63 -0
- reader_workbench/contracts/__init__.py +18 -0
- reader_workbench/contracts/builtins/__init__.py +36 -0
- reader_workbench/contracts/builtins/cytometry.py +140 -0
- reader_workbench/contracts/builtins/four_state_event_window.py +214 -0
- reader_workbench/contracts/builtins/generic.py +21 -0
- reader_workbench/contracts/builtins/logic.py +149 -0
- reader_workbench/contracts/builtins/plate_reader.py +47 -0
- reader_workbench/contracts/catalog.py +257 -0
- reader_workbench/contracts/model.py +109 -0
- reader_workbench/domains/__init__.py +1 -0
- reader_workbench/domains/cytometry/__init__.py +3 -0
- reader_workbench/domains/cytometry/analysis/__init__.py +29 -0
- reader_workbench/domains/cytometry/analysis/events.py +182 -0
- reader_workbench/domains/cytometry/analysis/gating.py +175 -0
- reader_workbench/domains/cytometry/analysis/workflow.py +248 -0
- reader_workbench/domains/cytometry/io/__init__.py +3 -0
- reader_workbench/domains/cytometry/io/fcs.py +135 -0
- reader_workbench/domains/cytometry/plots/__init__.py +5 -0
- reader_workbench/domains/cytometry/plots/diagnostic.py +155 -0
- reader_workbench/domains/logic/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/__init__.py +3 -0
- reader_workbench/domains/logic/crosstalk/pairs.py +661 -0
- reader_workbench/domains/logic/four_state_vector/__init__.py +7 -0
- reader_workbench/domains/logic/four_state_vector/builder.py +321 -0
- reader_workbench/domains/logic/four_state_vector/collection/__init__.py +20 -0
- reader_workbench/domains/logic/four_state_vector/collection/checks.py +49 -0
- reader_workbench/domains/logic/four_state_vector/collection/constants.py +29 -0
- reader_workbench/domains/logic/four_state_vector/collection/model.py +19 -0
- reader_workbench/domains/logic/four_state_vector/collection/render.py +383 -0
- reader_workbench/domains/logic/four_state_vector/collection/sources.py +185 -0
- reader_workbench/domains/logic/four_state_vector/config.py +214 -0
- reader_workbench/domains/logic/four_state_vector/diagnostic.py +361 -0
- reader_workbench/domains/logic/four_state_vector/heatmap.py +86 -0
- reader_workbench/domains/logic/four_state_vector/math.py +191 -0
- reader_workbench/domains/logic/four_state_vector/reference.py +85 -0
- reader_workbench/domains/logic/four_state_vector/selection.py +228 -0
- reader_workbench/domains/logic/four_state_vector/treatment_semantics.py +51 -0
- reader_workbench/domains/logic/four_state_vector/validation.py +19 -0
- reader_workbench/domains/logic/logic_symmetry/__init__.py +3 -0
- reader_workbench/domains/logic/logic_symmetry/encodings.py +93 -0
- reader_workbench/domains/logic/logic_symmetry/extract_corners.py +192 -0
- reader_workbench/domains/logic/logic_symmetry/main.py +236 -0
- reader_workbench/domains/logic/logic_symmetry/metrics.py +96 -0
- reader_workbench/domains/logic/logic_symmetry/overlay.py +129 -0
- reader_workbench/domains/logic/logic_symmetry/prep.py +138 -0
- reader_workbench/domains/logic/logic_symmetry/render.py +356 -0
- reader_workbench/domains/logic/treatment_columns.py +42 -0
- reader_workbench/domains/plate_reader/__init__.py +1 -0
- reader_workbench/domains/plate_reader/analysis/__init__.py +14 -0
- reader_workbench/domains/plate_reader/analysis/fold_change.py +474 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/__init__.py +21 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/aggregation.py +191 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contract_fields.py +50 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/contracts.py +320 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/design_dispositions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/disposition_records.py +140 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/event_sensitivity.py +27 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/materialize.py +274 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/observation_resampling.py +97 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/reduction.py +62 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/seeds.py +15 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/sources.py +308 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusion_validation.py +94 -0
- reader_workbench/domains/plate_reader/analysis/four_state_event_window/well_exclusions.py +54 -0
- reader_workbench/domains/plate_reader/analysis/timepoints.py +76 -0
- reader_workbench/domains/plate_reader/io/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/sample_map.py +65 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/__init__.py +6 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_kinetic.py +141 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_parser.py +295 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_shared.py +195 -0
- reader_workbench/domains/plate_reader/io/synergy_h1/_snapshot.py +130 -0
- reader_workbench/domains/plate_reader/ordering.py +59 -0
- reader_workbench/domains/plate_reader/plots/__init__.py +15 -0
- reader_workbench/domains/plate_reader/plots/_data.py +29 -0
- reader_workbench/domains/plate_reader/plots/common.py +346 -0
- reader_workbench/domains/plate_reader/plots/distributions.py +324 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych.py +525 -0
- reader_workbench/domains/plate_reader/plots/dual_reporter_triptych_render.py +195 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/__init__.py +26 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic.py +295 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_components.py +149 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_render.py +308 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/diagnostic_style.py +51 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/schema.py +8 -0
- reader_workbench/domains/plate_reader/plots/four_state_event_window/summary.py +140 -0
- reader_workbench/domains/plate_reader/plots/grouping.py +53 -0
- reader_workbench/domains/plate_reader/plots/panels/__init__.py +12 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot.py +161 -0
- reader_workbench/domains/plate_reader/plots/panels/snapshot_data.py +91 -0
- reader_workbench/domains/plate_reader/plots/panels/time_series.py +296 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic.py +435 -0
- reader_workbench/domains/plate_reader/plots/single_reporter_diagnostic_render.py +300 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/__init__.py +315 -0
- reader_workbench/domains/plate_reader/plots/snapshot_barplot/planning.py +168 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/__init__.py +205 -0
- reader_workbench/domains/plate_reader/plots/snapshot_heatmap/inputs.py +103 -0
- reader_workbench/domains/plate_reader/plots/time_series.py +317 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/__init__.py +479 -0
- reader_workbench/domains/plate_reader/plots/ts_and_snap/planning.py +283 -0
- reader_workbench/domains/time_series/__init__.py +29 -0
- reader_workbench/domains/time_series/aggregation.py +60 -0
- reader_workbench/domains/time_series/contracts.py +368 -0
- reader_workbench/domains/time_series/reduction.py +395 -0
- reader_workbench/errors.py +55 -0
- reader_workbench/maintenance/__init__.py +6 -0
- reader_workbench/maintenance/docs.py +335 -0
- reader_workbench/maintenance/model.py +28 -0
- reader_workbench/maintenance/release.py +39 -0
- reader_workbench/maintenance/skills.py +124 -0
- reader_workbench/plotting/__init__.py +20 -0
- reader_workbench/plotting/mpl.py +56 -0
- reader_workbench/plotting/sinks.py +69 -0
- reader_workbench/plotting/style.py +175 -0
- reader_workbench/plotting/utils.py +27 -0
- reader_workbench/plugins/__init__.py +1 -0
- reader_workbench/plugins/catalog.py +33 -0
- reader_workbench/plugins/export/__init__.py +0 -0
- reader_workbench/plugins/export/_paths.py +21 -0
- reader_workbench/plugins/export/csv.py +41 -0
- reader_workbench/plugins/export/xlsx.py +44 -0
- reader_workbench/plugins/ingest/__init__.py +0 -0
- reader_workbench/plugins/ingest/_discovery.py +58 -0
- reader_workbench/plugins/ingest/discovery_policy.py +66 -0
- reader_workbench/plugins/ingest/flow_cytometer.py +139 -0
- reader_workbench/plugins/ingest/synergy_h1.py +234 -0
- reader_workbench/plugins/manifests/__init__.py +1 -0
- reader_workbench/plugins/manifests/export.py +29 -0
- reader_workbench/plugins/manifests/ingest.py +29 -0
- reader_workbench/plugins/manifests/plot.py +161 -0
- reader_workbench/plugins/manifests/transform.py +172 -0
- reader_workbench/plugins/manifests/validator.py +18 -0
- reader_workbench/plugins/plot/__init__.py +0 -0
- reader_workbench/plugins/plot/_shared.py +55 -0
- reader_workbench/plugins/plot/cytometry_diagnostic.py +54 -0
- reader_workbench/plugins/plot/distributions.py +58 -0
- reader_workbench/plugins/plot/dual_reporter_triptych.py +224 -0
- reader_workbench/plugins/plot/four_state_event_window_diagnostic.py +133 -0
- reader_workbench/plugins/plot/four_state_event_window_summary.py +53 -0
- reader_workbench/plugins/plot/four_state_vector_collection.py +45 -0
- reader_workbench/plugins/plot/four_state_vector_diagnostic.py +105 -0
- reader_workbench/plugins/plot/four_state_vector_heatmap.py +63 -0
- reader_workbench/plugins/plot/logic_symmetry.py +56 -0
- reader_workbench/plugins/plot/single_reporter_diagnostic.py +279 -0
- reader_workbench/plugins/plot/snapshot_barplot.py +65 -0
- reader_workbench/plugins/plot/snapshot_heatmap.py +104 -0
- reader_workbench/plugins/plot/time_series.py +114 -0
- reader_workbench/plugins/plot/ts_and_snap.py +210 -0
- reader_workbench/plugins/transform/__init__.py +0 -0
- reader_workbench/plugins/transform/_four_state_vector.py +204 -0
- reader_workbench/plugins/transform/_labeling.py +109 -0
- reader_workbench/plugins/transform/alias.py +70 -0
- reader_workbench/plugins/transform/assay_labels.py +62 -0
- reader_workbench/plugins/transform/blank.py +79 -0
- reader_workbench/plugins/transform/crosstalk_pairs.py +180 -0
- reader_workbench/plugins/transform/cytometry_gating.py +120 -0
- reader_workbench/plugins/transform/fold_change.py +79 -0
- reader_workbench/plugins/transform/four_state_event_window.py +93 -0
- reader_workbench/plugins/transform/four_state_vector.py +62 -0
- reader_workbench/plugins/transform/four_state_vector_collection.py +41 -0
- reader_workbench/plugins/transform/logic_symmetry.py +67 -0
- reader_workbench/plugins/transform/outlier_filter.py +60 -0
- reader_workbench/plugins/transform/overflow.py +197 -0
- reader_workbench/plugins/transform/ratio.py +237 -0
- reader_workbench/plugins/transform/sample_map.py +170 -0
- reader_workbench/plugins/transform/sample_metadata.py +94 -0
- reader_workbench/plugins/validator/__init__.py +1 -0
- reader_workbench/plugins/validator/to_tidy_plus_map.py +155 -0
- reader_workbench/protocols/__init__.py +80 -0
- reader_workbench/protocols/_builtins_plate_reader_growth.py +179 -0
- reader_workbench/protocols/_builtins_plate_reader_variants.py +274 -0
- reader_workbench/protocols/builtins.py +1656 -0
- reader_workbench/protocols/compiler.py +22 -0
- reader_workbench/protocols/compilers/__init__.py +1 -0
- reader_workbench/protocols/compilers/common.py +100 -0
- reader_workbench/protocols/compilers/cytometry.py +87 -0
- reader_workbench/protocols/compilers/generic.py +14 -0
- reader_workbench/protocols/compilers/logic.py +245 -0
- reader_workbench/protocols/compilers/plate_reader.py +937 -0
- reader_workbench/protocols/compilers/plate_reader_pipeline.py +197 -0
- reader_workbench/protocols/model.py +1486 -0
- reader_workbench/protocols/semantic_coverage.py +234 -0
- reader_workbench/runtime/__init__.py +12 -0
- reader_workbench/runtime/builtin.py +23 -0
- reader_workbench/runtime/model.py +42 -0
- reader_workbench/workbench/__init__.py +60 -0
- reader_workbench/workbench/assets/__init__.py +22 -0
- reader_workbench/workbench/assets/types.py +118 -0
- reader_workbench/workbench/audit/__init__.py +5 -0
- reader_workbench/workbench/audit/experiments.py +307 -0
- reader_workbench/workbench/audit/staging.py +187 -0
- reader_workbench/workbench/cli/__init__.py +51 -0
- reader_workbench/workbench/cli/_lazy.py +9 -0
- reader_workbench/workbench/cli/_records_view.py +150 -0
- reader_workbench/workbench/cli/_surface_execution.py +443 -0
- reader_workbench/workbench/cli/audit.py +95 -0
- reader_workbench/workbench/cli/automation.py +229 -0
- reader_workbench/workbench/cli/demo.py +46 -0
- reader_workbench/workbench/cli/dop.py +91 -0
- reader_workbench/workbench/cli/experiments.py +635 -0
- reader_workbench/workbench/cli/helpers.py +232 -0
- reader_workbench/workbench/cli/main.py +59 -0
- reader_workbench/workbench/cli/maintenance.py +82 -0
- reader_workbench/workbench/cli/notebooks.py +260 -0
- reader_workbench/workbench/cli/pagination.py +117 -0
- reader_workbench/workbench/cli/protocols.py +336 -0
- reader_workbench/workbench/cli/shared.py +309 -0
- reader_workbench/workbench/cli/surfaces.py +534 -0
- reader_workbench/workbench/cli/verification.py +128 -0
- reader_workbench/workbench/commands.py +10 -0
- reader_workbench/workbench/config/__init__.py +47 -0
- reader_workbench/workbench/config/identity.py +13 -0
- reader_workbench/workbench/config/load.py +405 -0
- reader_workbench/workbench/config/model.py +274 -0
- reader_workbench/workbench/context.py +26 -0
- reader_workbench/workbench/decl/__init__.py +31 -0
- reader_workbench/workbench/decl/build.py +190 -0
- reader_workbench/workbench/decl/model.py +81 -0
- reader_workbench/workbench/dop/__init__.py +12 -0
- reader_workbench/workbench/dop/builtins.py +261 -0
- reader_workbench/workbench/dop/model.py +209 -0
- reader_workbench/workbench/engine/__init__.py +42 -0
- reader_workbench/workbench/engine/_shared.py +76 -0
- reader_workbench/workbench/engine/contracts.py +283 -0
- reader_workbench/workbench/engine/execution.py +326 -0
- reader_workbench/workbench/engine/file_outputs.py +260 -0
- reader_workbench/workbench/engine/inputs.py +161 -0
- reader_workbench/workbench/engine/invocations.py +507 -0
- reader_workbench/workbench/engine/planning.py +72 -0
- reader_workbench/workbench/engine/runtime.py +464 -0
- reader_workbench/workbench/engine/setup.py +149 -0
- reader_workbench/workbench/engine/validation.py +684 -0
- reader_workbench/workbench/experiment/__init__.py +47 -0
- reader_workbench/workbench/experiment/model.py +381 -0
- reader_workbench/workbench/experiments.py +133 -0
- reader_workbench/workbench/graph/__init__.py +47 -0
- reader_workbench/workbench/graph/nodes.py +102 -0
- reader_workbench/workbench/graph/normalize.py +177 -0
- reader_workbench/workbench/graph/refs.py +148 -0
- reader_workbench/workbench/input_discovery.py +19 -0
- reader_workbench/workbench/inspection/__init__.py +3 -0
- reader_workbench/workbench/inspection/catalogs.py +128 -0
- reader_workbench/workbench/inspection/common.py +92 -0
- reader_workbench/workbench/inspection/dop.py +64 -0
- reader_workbench/workbench/inspection/experiments.py +449 -0
- reader_workbench/workbench/inspection/inventory.py +68 -0
- reader_workbench/workbench/inspection/protocols.py +368 -0
- reader_workbench/workbench/inspection/readiness.py +333 -0
- reader_workbench/workbench/inspection/reports.py +367 -0
- reader_workbench/workbench/inspection/results.py +166 -0
- reader_workbench/workbench/inspection/runtime.py +287 -0
- reader_workbench/workbench/inspection/semantics.py +192 -0
- reader_workbench/workbench/inspection/validation.py +30 -0
- reader_workbench/workbench/notebooks/__init__.py +17 -0
- reader_workbench/workbench/notebooks/_launch_registry.py +112 -0
- reader_workbench/workbench/notebooks/_launch_runtime.py +104 -0
- reader_workbench/workbench/notebooks/components/__init__.py +21 -0
- reader_workbench/workbench/notebooks/components/deliverables.py +403 -0
- reader_workbench/workbench/notebooks/components/overview.py +119 -0
- reader_workbench/workbench/notebooks/eda.marimo.py.txt +153 -0
- reader_workbench/workbench/notebooks/launch.py +274 -0
- reader_workbench/workbench/notebooks/presentation.py +136 -0
- reader_workbench/workbench/notebooks/scaffold.py +60 -0
- reader_workbench/workbench/ontology.py +78 -0
- reader_workbench/workbench/paths.py +44 -0
- reader_workbench/workbench/ports/__init__.py +31 -0
- reader_workbench/workbench/ports/model.py +168 -0
- reader_workbench/workbench/records/__init__.py +44 -0
- reader_workbench/workbench/records/epoch.py +329 -0
- reader_workbench/workbench/records/evidence.py +247 -0
- reader_workbench/workbench/records/identity.py +87 -0
- reader_workbench/workbench/records/locking.py +185 -0
- reader_workbench/workbench/records/model.py +711 -0
- reader_workbench/workbench/records/sources.py +73 -0
- reader_workbench/workbench/records/store.py +1022 -0
- reader_workbench/workbench/records/verification.py +998 -0
- reader_workbench/workbench/registry.py +333 -0
- reader_workbench/workbench/spec_overrides.py +215 -0
- reader_workbench-1.0.0.dist-info/METADATA +91 -0
- reader_workbench-1.0.0.dist-info/RECORD +293 -0
- reader_workbench-1.0.0.dist-info/WHEEL +5 -0
- reader_workbench-1.0.0.dist-info/entry_points.txt +2 -0
- reader_workbench-1.0.0.dist-info/licenses/LICENSE +21 -0
- reader_workbench-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Four-state logic-intensity measurement vector."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Literal
|
|
6
|
+
|
|
7
|
+
from pydantic import Field
|
|
8
|
+
|
|
9
|
+
from reader_workbench.plugins.transform._four_state_vector import (
|
|
10
|
+
build_four_state_vector_plugin_result,
|
|
11
|
+
log_four_state_vector_plugin_result,
|
|
12
|
+
)
|
|
13
|
+
from reader_workbench.workbench.ports import dataframe_input, dataframe_output
|
|
14
|
+
from reader_workbench.workbench.registry import Plugin, PluginConfig
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class FourStateVectorResponseBinding(PluginConfig):
|
|
18
|
+
logic_channel: str = Field(min_length=1)
|
|
19
|
+
intensity_channel: str = Field(min_length=1)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class FourStateVectorReferenceBinding(PluginConfig):
|
|
23
|
+
design_id: str = Field(min_length=1)
|
|
24
|
+
observation_stat: Literal["mean", "median"] = "mean"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class FourStateVectorCfg(PluginConfig):
|
|
28
|
+
response: FourStateVectorResponseBinding
|
|
29
|
+
design_by: list[str] = Field(default_factory=lambda: ["design_id"])
|
|
30
|
+
time_column: str = "time"
|
|
31
|
+
treatment_column: str | None = None
|
|
32
|
+
time_mode: Literal["nearest", "last_before", "first_after", "exact"] = "nearest"
|
|
33
|
+
target_time_h: float | None = None
|
|
34
|
+
time_tolerance_h: float | None = 0.5
|
|
35
|
+
state_map_ref: str = Field(min_length=1)
|
|
36
|
+
reference: FourStateVectorReferenceBinding
|
|
37
|
+
require_all_corners_per_design: bool = True
|
|
38
|
+
eps_ratio: float = 1e-9
|
|
39
|
+
eps_range: float = 1e-12
|
|
40
|
+
eps_ref: float = 1e-9
|
|
41
|
+
eps_abs: float = 0.0
|
|
42
|
+
ref_add_alpha: float = 0.0
|
|
43
|
+
log2_offset_delta: float = 0.0
|
|
44
|
+
exclude_reference_from_output: bool = True
|
|
45
|
+
carry_metadata: list[str] = Field(default_factory=lambda: ["sequence", "id"])
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class FourStateVectorTransform(Plugin):
|
|
49
|
+
ConfigModel = FourStateVectorCfg
|
|
50
|
+
|
|
51
|
+
@classmethod
|
|
52
|
+
def input_ports(cls):
|
|
53
|
+
return {"df": dataframe_input("df", "plate_reader.annotated.v1")}
|
|
54
|
+
|
|
55
|
+
@classmethod
|
|
56
|
+
def output_ports(cls):
|
|
57
|
+
return {"vector": dataframe_output("vector", "logic.four_state_vector.v1")}
|
|
58
|
+
|
|
59
|
+
def run(self, ctx, inputs, cfg: FourStateVectorCfg):
|
|
60
|
+
result = build_four_state_vector_plugin_result(ctx=ctx, df=inputs["df"], cfg=cfg)
|
|
61
|
+
log_four_state_vector_plugin_result(ctx=ctx, result=result)
|
|
62
|
+
return {"vector": result.vector}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from reader_workbench.domains.logic.four_state_vector.collection import (
|
|
4
|
+
FourStateVectorSource,
|
|
5
|
+
collect_four_state_vector_sources,
|
|
6
|
+
)
|
|
7
|
+
from reader_workbench.workbench.ports import dataframe_output, record_collection_input
|
|
8
|
+
from reader_workbench.workbench.records import SourceRecordCollection
|
|
9
|
+
from reader_workbench.workbench.registry import Plugin, PluginConfig
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class FourStateVectorCollectionCfg(PluginConfig):
|
|
13
|
+
pass
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class FourStateVectorCollectionTransform(Plugin):
|
|
17
|
+
"""Collect exact vector revisions; workspace discovery stays in Reader core."""
|
|
18
|
+
|
|
19
|
+
ConfigModel = FourStateVectorCollectionCfg
|
|
20
|
+
|
|
21
|
+
@classmethod
|
|
22
|
+
def input_ports(cls):
|
|
23
|
+
return {"sources": record_collection_input("sources", "logic.four_state_vector.v1")}
|
|
24
|
+
|
|
25
|
+
@classmethod
|
|
26
|
+
def output_ports(cls):
|
|
27
|
+
return {"vectors": dataframe_output("vectors", "logic.four_state_vector_collection.v1")}
|
|
28
|
+
|
|
29
|
+
def run(self, ctx, inputs, cfg):
|
|
30
|
+
collection: SourceRecordCollection = inputs["sources"]
|
|
31
|
+
sources = tuple(
|
|
32
|
+
FourStateVectorSource(
|
|
33
|
+
resource_id=item.ref.resource_id,
|
|
34
|
+
experiment_id=item.ref.experiment_id,
|
|
35
|
+
record_id=item.ref.record_id,
|
|
36
|
+
revision_digest=item.revision_digest,
|
|
37
|
+
frame=item.load_dataframe(),
|
|
38
|
+
)
|
|
39
|
+
for item in collection
|
|
40
|
+
)
|
|
41
|
+
return {"vectors": collect_four_state_vector_sources(sources).frame}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Literal
|
|
4
|
+
|
|
5
|
+
from pydantic import Field
|
|
6
|
+
|
|
7
|
+
from reader_workbench.workbench.ports import dataframe_input, dataframe_output
|
|
8
|
+
from reader_workbench.workbench.registry import Plugin, PluginConfig
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class LogicSymmetryPrepCfg(PluginConfig):
|
|
12
|
+
enable: bool = False
|
|
13
|
+
mode: Literal["first", "last", "median", "exact", "nearest"] = "last"
|
|
14
|
+
target_time: float | None = None
|
|
15
|
+
tolerance: float = Field(0.51, ge=0)
|
|
16
|
+
align_corners: bool = False
|
|
17
|
+
case_sensitive_treatments: bool | None = None
|
|
18
|
+
time_column: str = "time"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class LogicSymmetryCfg(PluginConfig):
|
|
22
|
+
response_channel: str = Field(min_length=1)
|
|
23
|
+
design_by: list[str] = Field(default_factory=lambda: ["design_id"], min_length=1)
|
|
24
|
+
batch_col: str = Field("batch", min_length=1)
|
|
25
|
+
treatment_column: str | None = None
|
|
26
|
+
state_map_ref: str = Field(min_length=1)
|
|
27
|
+
observation_stat: Literal["mean", "median"] = "mean"
|
|
28
|
+
prep: LogicSymmetryPrepCfg = Field(default_factory=LogicSymmetryPrepCfg)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class LogicSymmetryTransform(Plugin):
|
|
32
|
+
"""Materialize logic-symmetry metrics as a normal pipeline record."""
|
|
33
|
+
|
|
34
|
+
ConfigModel = LogicSymmetryCfg
|
|
35
|
+
|
|
36
|
+
@classmethod
|
|
37
|
+
def input_ports(cls):
|
|
38
|
+
return {"df": dataframe_input("df", "plate_reader.annotated.v1")}
|
|
39
|
+
|
|
40
|
+
@classmethod
|
|
41
|
+
def output_ports(cls):
|
|
42
|
+
return {"table": dataframe_output("table", "logic_symmetry.v1")}
|
|
43
|
+
|
|
44
|
+
def run(self, ctx, inputs, cfg: LogicSymmetryCfg):
|
|
45
|
+
if ctx.experiment is None:
|
|
46
|
+
raise ValueError("logic_symmetry requires experiment semantics in the run context")
|
|
47
|
+
state_space = ctx.experiment.annotations.resolve_ordered_state_space(ref=cfg.state_map_ref)
|
|
48
|
+
if state_space.state_ids != ("00", "10", "01", "11"):
|
|
49
|
+
raise ValueError("Logic-symmetry state space must declare exactly 00, 10, 01, 11 in that order")
|
|
50
|
+
|
|
51
|
+
from reader_workbench.domains.logic.logic_symmetry import summarize_logic_symmetry # noqa: PLC0415
|
|
52
|
+
|
|
53
|
+
prep = cfg.prep.model_dump()
|
|
54
|
+
if prep["case_sensitive_treatments"] is None:
|
|
55
|
+
prep["case_sensitive_treatments"] = state_space.case_sensitive
|
|
56
|
+
table = summarize_logic_symmetry(
|
|
57
|
+
inputs["df"],
|
|
58
|
+
response_channel=cfg.response_channel,
|
|
59
|
+
design_by=cfg.design_by,
|
|
60
|
+
batch_col=cfg.batch_col,
|
|
61
|
+
treatment_column=cfg.treatment_column or state_space.column,
|
|
62
|
+
treatment_map=dict(state_space.source_values),
|
|
63
|
+
treatment_case_sensitive=state_space.case_sensitive,
|
|
64
|
+
observation_stat=cfg.observation_stat,
|
|
65
|
+
prep=prep,
|
|
66
|
+
)
|
|
67
|
+
return {"table": table}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Simple z-score filter per (channel, time)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
|
|
8
|
+
from reader_workbench.workbench.ports import dataframe_input, dataframe_output
|
|
9
|
+
from reader_workbench.workbench.registry import Plugin, PluginConfig
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class OutlierCfg(PluginConfig):
|
|
13
|
+
enable: bool = False
|
|
14
|
+
z_thresh: float = 4.0
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class OutlierFilter(Plugin):
|
|
18
|
+
ConfigModel = OutlierCfg
|
|
19
|
+
|
|
20
|
+
@classmethod
|
|
21
|
+
def input_ports(cls):
|
|
22
|
+
return {"df": dataframe_input("df", "tidy.v1")}
|
|
23
|
+
|
|
24
|
+
@classmethod
|
|
25
|
+
def output_ports(cls):
|
|
26
|
+
return cls.passthrough_output_ports(
|
|
27
|
+
outputs={"df": dataframe_output("df", "tidy.v1")},
|
|
28
|
+
passthrough={"df": "df"},
|
|
29
|
+
promoted_examples={"df": ("plate_reader.annotated.v1",)},
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
def resolve_output_ports(self, *, inputs, outputs, cfg, where):
|
|
33
|
+
del cfg
|
|
34
|
+
return self.inherit_dataframe_output_ports(
|
|
35
|
+
inputs=inputs,
|
|
36
|
+
outputs=outputs,
|
|
37
|
+
passthrough={"df": "df"},
|
|
38
|
+
where=where,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
def run(self, ctx, inputs, cfg: OutlierCfg):
|
|
42
|
+
if not cfg.enable:
|
|
43
|
+
return {"df": inputs["df"].copy()}
|
|
44
|
+
|
|
45
|
+
df = inputs["df"].copy()
|
|
46
|
+
df["value"] = pd.to_numeric(df["value"], errors="coerce")
|
|
47
|
+
|
|
48
|
+
def _f(g: pd.DataFrame) -> pd.DataFrame:
|
|
49
|
+
s = g["value"].dropna()
|
|
50
|
+
if s.size <= 1:
|
|
51
|
+
return g
|
|
52
|
+
mu = float(s.mean())
|
|
53
|
+
sd = float(s.std(ddof=1)) if s.size > 1 else 0.0
|
|
54
|
+
if not np.isfinite(sd) or sd <= 0:
|
|
55
|
+
return g
|
|
56
|
+
z = (g["value"] - mu) / sd
|
|
57
|
+
return g.loc[z.abs() <= float(cfg.z_thresh)]
|
|
58
|
+
|
|
59
|
+
out = df.groupby(["channel", "time"], group_keys=False).apply(_f)
|
|
60
|
+
return {"df": out}
|
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Mapping
|
|
4
|
+
from typing import Literal
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
|
|
9
|
+
from reader_workbench.workbench.ports import dataframe_input, dataframe_output
|
|
10
|
+
from reader_workbench.workbench.registry import Plugin, PluginConfig
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class OverflowCfg(PluginConfig):
|
|
14
|
+
action: Literal["max", "drop", "nan", "none"] = "max"
|
|
15
|
+
clip_quantile: float = 0.999
|
|
16
|
+
# New: explicit capping strategy
|
|
17
|
+
cap_strategy: Literal["provided", "infer", "quantile"] = "quantile"
|
|
18
|
+
per_channel_caps: Mapping[str, float] | None = None
|
|
19
|
+
# New: how to detect overflow rows
|
|
20
|
+
flag_column: str = "overflow"
|
|
21
|
+
treat_inf_as_overflow: bool = True
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class OverflowHandling(Plugin):
|
|
25
|
+
ConfigModel = OverflowCfg
|
|
26
|
+
|
|
27
|
+
@classmethod
|
|
28
|
+
def input_ports(cls):
|
|
29
|
+
return {"df": dataframe_input("df", "tidy.v1")}
|
|
30
|
+
|
|
31
|
+
@classmethod
|
|
32
|
+
def output_ports(cls):
|
|
33
|
+
return cls.passthrough_output_ports(
|
|
34
|
+
outputs={"df": dataframe_output("df", "tidy.v1")},
|
|
35
|
+
passthrough={"df": "df"},
|
|
36
|
+
promoted_examples={"df": ("plate_reader.annotated.v1",)},
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
def resolve_output_ports(self, *, inputs, outputs, cfg, where):
|
|
40
|
+
del cfg
|
|
41
|
+
return self.inherit_dataframe_output_ports(
|
|
42
|
+
inputs=inputs,
|
|
43
|
+
outputs=outputs,
|
|
44
|
+
passthrough={"df": "df"},
|
|
45
|
+
where=where,
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
def run(self, ctx, inputs, cfg: OverflowCfg):
|
|
49
|
+
df = inputs["df"].copy()
|
|
50
|
+
df["value"] = pd.to_numeric(df["value"], errors="coerce")
|
|
51
|
+
act = cfg.action.lower()
|
|
52
|
+
if act == "none":
|
|
53
|
+
unexpected_nonfinite = df["value"].isna() | np.isneginf(df["value"])
|
|
54
|
+
if unexpected_nonfinite.any():
|
|
55
|
+
raise ValueError("overflow_handling: NaN or negative infinity cannot represent instrument overflow")
|
|
56
|
+
policy_clipped, prior_overflow, bounds = _incoming_value_provenance(df)
|
|
57
|
+
flagged = _declared_instrument_overflow(df, flag_column=cfg.flag_column)
|
|
58
|
+
if cfg.treat_inf_as_overflow:
|
|
59
|
+
flagged = flagged | np.isposinf(df["value"])
|
|
60
|
+
elif (~np.isfinite(df["value"]) & ~(flagged | prior_overflow)).any():
|
|
61
|
+
raise ValueError("overflow_handling: non-finite values must be classified as instrument overflow")
|
|
62
|
+
instrument_overflow = prior_overflow | flagged
|
|
63
|
+
bounds = bounds.copy()
|
|
64
|
+
bounds.loc[flagged] = bounds.loc[flagged].map(_union_lower_bound)
|
|
65
|
+
df["value_policy_clipped"] = policy_clipped
|
|
66
|
+
df["value_instrument_overflow"] = instrument_overflow
|
|
67
|
+
df["value_bound_kind"] = bounds
|
|
68
|
+
df[cfg.flag_column] = instrument_overflow
|
|
69
|
+
return {"df": df}
|
|
70
|
+
if act == "drop":
|
|
71
|
+
unexpected_nonfinite = df["value"].isna() | np.isneginf(df["value"])
|
|
72
|
+
if unexpected_nonfinite.any():
|
|
73
|
+
raise ValueError("overflow_handling: NaN or negative infinity cannot represent instrument overflow")
|
|
74
|
+
policy_clipped, prior_overflow, bounds = _incoming_value_provenance(df)
|
|
75
|
+
flagged = _declared_instrument_overflow(df, flag_column=cfg.flag_column)
|
|
76
|
+
if cfg.treat_inf_as_overflow:
|
|
77
|
+
flagged = flagged | np.isposinf(df["value"])
|
|
78
|
+
elif (~np.isfinite(df["value"]) & ~(flagged | prior_overflow)).any():
|
|
79
|
+
raise ValueError("overflow_handling: non-finite values must be classified as instrument overflow")
|
|
80
|
+
drop_rows = policy_clipped | prior_overflow | bounds.ne("exact") | flagged
|
|
81
|
+
out = df.loc[~drop_rows].copy()
|
|
82
|
+
out["value_policy_clipped"] = False
|
|
83
|
+
out["value_instrument_overflow"] = False
|
|
84
|
+
out["value_bound_kind"] = "exact"
|
|
85
|
+
out[cfg.flag_column] = False
|
|
86
|
+
return {"df": out}
|
|
87
|
+
if act == "nan":
|
|
88
|
+
return {"df": df}
|
|
89
|
+
if act == "max":
|
|
90
|
+
# 1) mark which rows are overflowed
|
|
91
|
+
flagged = _declared_instrument_overflow(df, flag_column=cfg.flag_column)
|
|
92
|
+
if cfg.treat_inf_as_overflow:
|
|
93
|
+
flagged = flagged | ~np.isfinite(df["value"])
|
|
94
|
+
elif (~np.isfinite(df["value"]) & ~flagged).any():
|
|
95
|
+
raise ValueError("overflow_handling: non-finite values must be classified as instrument overflow")
|
|
96
|
+
|
|
97
|
+
# 2) compute per-channel caps explicitly
|
|
98
|
+
if cfg.cap_strategy == "provided":
|
|
99
|
+
if not cfg.per_channel_caps:
|
|
100
|
+
raise ValueError("overflow_handling: cap_strategy='provided' but per_channel_caps is empty")
|
|
101
|
+
caps = pd.Series({str(k): float(v) for k, v in cfg.per_channel_caps.items()}, name="__cap__")
|
|
102
|
+
elif cfg.cap_strategy == "infer":
|
|
103
|
+
base = df[np.isfinite(df["value"])]
|
|
104
|
+
if base.empty:
|
|
105
|
+
raise ValueError("overflow_handling: cap_strategy='infer' but no finite values available")
|
|
106
|
+
caps = base.groupby("channel")["value"].max().rename("__cap__")
|
|
107
|
+
elif cfg.cap_strategy == "quantile":
|
|
108
|
+
base = df[np.isfinite(df["value"])]
|
|
109
|
+
if base.empty:
|
|
110
|
+
raise ValueError("overflow_handling: cap_strategy='quantile' but no finite values available")
|
|
111
|
+
caps = base.groupby("channel")["value"].quantile(float(cfg.clip_quantile)).rename("__cap__")
|
|
112
|
+
else:
|
|
113
|
+
raise ValueError(f"overflow_handling: unknown cap_strategy {cfg.cap_strategy!r}")
|
|
114
|
+
|
|
115
|
+
out = df.join(caps, on="channel")
|
|
116
|
+
if out["__cap__"].isna().any():
|
|
117
|
+
missing = sorted(out.loc[out["__cap__"].isna(), "channel"].astype(str).unique())
|
|
118
|
+
raise ValueError(f"overflow_handling: missing cap for channels: {missing}")
|
|
119
|
+
|
|
120
|
+
# 3) preserve why an observation is no longer exact before clamping.
|
|
121
|
+
# Explicit instrument overflow and finite policy clipping are different
|
|
122
|
+
# evidence states even though both land on the configured upper cap.
|
|
123
|
+
policy_clipped = np.isfinite(out["value"]) & out["value"].gt(out["__cap__"]) & ~flagged
|
|
124
|
+
out["value_policy_clipped"] = policy_clipped.astype(bool)
|
|
125
|
+
out["value_instrument_overflow"] = flagged.astype(bool)
|
|
126
|
+
out["value_bound_kind"] = np.where(policy_clipped | flagged, "lower", "exact")
|
|
127
|
+
out[cfg.flag_column] = flagged.astype(bool)
|
|
128
|
+
|
|
129
|
+
# 4) clamp everything to the cap; overflowed rows land exactly on the cap
|
|
130
|
+
out.loc[flagged, "value"] = np.inf # ensure clamp hits the cap deterministically
|
|
131
|
+
out["value"] = np.minimum(out["value"], out["__cap__"])
|
|
132
|
+
|
|
133
|
+
# 5) concise log
|
|
134
|
+
if ctx.logger is not None:
|
|
135
|
+
policy_counts = policy_clipped.groupby(out["channel"]).sum().astype(int)
|
|
136
|
+
overflow_counts = flagged.groupby(out["channel"]).sum().astype(int)
|
|
137
|
+
ctx.logger.info(
|
|
138
|
+
"overflow_handling • strategy=%s • policy_clipped_rows=%d • "
|
|
139
|
+
"instrument_overflow_rows=%d • policy_by_channel=%s • instrument_by_channel=%s",
|
|
140
|
+
cfg.cap_strategy,
|
|
141
|
+
int(policy_clipped.sum()),
|
|
142
|
+
int(flagged.sum()),
|
|
143
|
+
dict(policy_counts[policy_counts > 0]),
|
|
144
|
+
dict(overflow_counts[overflow_counts > 0]),
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
return {"df": out.drop(columns="__cap__")}
|
|
148
|
+
raise ValueError(f"unknown overflow action {cfg.action}")
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _incoming_value_provenance(frame: pd.DataFrame) -> tuple[pd.Series, pd.Series, pd.Series]:
|
|
152
|
+
fields = {"value_policy_clipped", "value_instrument_overflow", "value_bound_kind"}
|
|
153
|
+
present = fields & set(frame.columns)
|
|
154
|
+
if present and present != fields:
|
|
155
|
+
raise ValueError("overflow_handling: value provenance must provide all three fields together")
|
|
156
|
+
if not present:
|
|
157
|
+
return (
|
|
158
|
+
pd.Series(False, index=frame.index, dtype=bool),
|
|
159
|
+
pd.Series(False, index=frame.index, dtype=bool),
|
|
160
|
+
pd.Series("exact", index=frame.index, dtype=object),
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
policy_clipped = _strict_provenance_boolean(frame["value_policy_clipped"], field="value_policy_clipped")
|
|
164
|
+
instrument_overflow = _strict_provenance_boolean(
|
|
165
|
+
frame["value_instrument_overflow"], field="value_instrument_overflow"
|
|
166
|
+
)
|
|
167
|
+
bounds = frame["value_bound_kind"]
|
|
168
|
+
if bounds.isna().any() or not bounds.map(lambda value: isinstance(value, str)).all():
|
|
169
|
+
raise ValueError("overflow_handling: value_bound_kind provenance must contain strings without missing values")
|
|
170
|
+
bounds = bounds.astype(str)
|
|
171
|
+
allowed = {"exact", "lower", "upper", "indeterminate"}
|
|
172
|
+
unknown = sorted(set(bounds) - allowed)
|
|
173
|
+
if unknown:
|
|
174
|
+
raise ValueError(f"overflow_handling: unsupported value_bound_kind provenance: {unknown}")
|
|
175
|
+
if not (policy_clipped | instrument_overflow).eq(bounds.ne("exact")).all():
|
|
176
|
+
raise ValueError("overflow_handling: clipping and overflow provenance disagrees with value_bound_kind")
|
|
177
|
+
return policy_clipped, instrument_overflow, bounds
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _declared_instrument_overflow(frame: pd.DataFrame, *, flag_column: str) -> pd.Series:
|
|
181
|
+
flagged = pd.Series(False, index=frame.index, dtype=bool)
|
|
182
|
+
if flag_column not in frame.columns:
|
|
183
|
+
return flagged
|
|
184
|
+
raw_flags = frame[flag_column]
|
|
185
|
+
if raw_flags.isna().any() or not raw_flags.map(lambda value: isinstance(value, (bool, np.bool_))).all():
|
|
186
|
+
raise ValueError(f"overflow_handling: {flag_column!r} must contain booleans without missing values")
|
|
187
|
+
return raw_flags.astype(bool)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _strict_provenance_boolean(values: pd.Series, *, field: str) -> pd.Series:
|
|
191
|
+
if values.isna().any() or not values.map(lambda value: isinstance(value, (bool, np.bool_))).all():
|
|
192
|
+
raise ValueError(f"overflow_handling: {field} provenance must contain booleans without missing values")
|
|
193
|
+
return values.astype(bool)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _union_lower_bound(bound: object) -> str:
|
|
197
|
+
return "lower" if str(bound) in {"exact", "lower"} else "indeterminate"
|
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from contextlib import suppress
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
from pydantic import Field
|
|
8
|
+
|
|
9
|
+
from reader_workbench.workbench.ports import dataframe_input, dataframe_output
|
|
10
|
+
from reader_workbench.workbench.registry import Plugin, PluginConfig
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class RatioCfg(PluginConfig):
|
|
14
|
+
name: str
|
|
15
|
+
numerator: str
|
|
16
|
+
denominator: str
|
|
17
|
+
align_on: list[str] = Field(default_factory=lambda: ["position", "time"])
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class RatioTransform(Plugin):
|
|
21
|
+
ConfigModel = RatioCfg
|
|
22
|
+
|
|
23
|
+
@classmethod
|
|
24
|
+
def input_ports(cls):
|
|
25
|
+
return {"df": dataframe_input("df", "tidy.v1")}
|
|
26
|
+
|
|
27
|
+
@classmethod
|
|
28
|
+
def output_ports(cls):
|
|
29
|
+
return cls.passthrough_output_ports(
|
|
30
|
+
outputs={"df": dataframe_output("df", "tidy.v1")},
|
|
31
|
+
passthrough={"df": "df"},
|
|
32
|
+
promoted_examples={"df": ("plate_reader.annotated.v1",)},
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
def resolve_output_ports(self, *, inputs, outputs, cfg, where):
|
|
36
|
+
del cfg
|
|
37
|
+
return self.inherit_dataframe_output_ports(
|
|
38
|
+
inputs=inputs,
|
|
39
|
+
outputs=outputs,
|
|
40
|
+
passthrough={"df": "df"},
|
|
41
|
+
where=where,
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
def run(self, ctx, inputs, cfg: RatioCfg):
|
|
45
|
+
input_columns = list(inputs["df"].columns)
|
|
46
|
+
df, emit_value_provenance = _with_value_provenance(inputs["df"])
|
|
47
|
+
|
|
48
|
+
# Build alignment key; auto-augment with per-sheet/scope cols if present
|
|
49
|
+
key = [c for c in cfg.align_on if c in df.columns]
|
|
50
|
+
for extra in ("sheet_index", "sheet_name", "source"):
|
|
51
|
+
if extra in df.columns and extra not in key:
|
|
52
|
+
key.append(extra)
|
|
53
|
+
if not key:
|
|
54
|
+
available = sorted(set(df.columns))
|
|
55
|
+
raise ValueError(
|
|
56
|
+
"ratio: none of align_on columns are present in the input.\n"
|
|
57
|
+
f" align_on: {cfg.align_on}\n"
|
|
58
|
+
f" available: {available}"
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
# Partition numerator/denominator; keep ALL metadata on numerator side
|
|
62
|
+
lhs = (
|
|
63
|
+
df[df["channel"] == cfg.numerator]
|
|
64
|
+
.rename(
|
|
65
|
+
columns={
|
|
66
|
+
"value": "__num__",
|
|
67
|
+
"value_policy_clipped": "__num_policy_clipped__",
|
|
68
|
+
"value_instrument_overflow": "__num_instrument_overflow__",
|
|
69
|
+
"value_bound_kind": "__num_bound_kind__",
|
|
70
|
+
}
|
|
71
|
+
)
|
|
72
|
+
.copy()
|
|
73
|
+
)
|
|
74
|
+
rhs = (
|
|
75
|
+
df[df["channel"] == cfg.denominator]
|
|
76
|
+
.rename(
|
|
77
|
+
columns={
|
|
78
|
+
"value": "__den__",
|
|
79
|
+
"value_policy_clipped": "__den_policy_clipped__",
|
|
80
|
+
"value_instrument_overflow": "__den_instrument_overflow__",
|
|
81
|
+
"value_bound_kind": "__den_bound_kind__",
|
|
82
|
+
}
|
|
83
|
+
)
|
|
84
|
+
.copy()
|
|
85
|
+
)
|
|
86
|
+
if lhs.empty or rhs.empty:
|
|
87
|
+
available = sorted(df["channel"].dropna().astype(str).unique().tolist())
|
|
88
|
+
missing = []
|
|
89
|
+
if lhs.empty:
|
|
90
|
+
missing.append(cfg.numerator)
|
|
91
|
+
if rhs.empty:
|
|
92
|
+
missing.append(cfg.denominator)
|
|
93
|
+
raise ValueError(
|
|
94
|
+
f"ratio: requested channel(s) missing from input.\n missing: {missing}\n available: {available}"
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
# Keep only join keys + denominator on RHS to avoid suffix collisions
|
|
98
|
+
rhs = rhs[
|
|
99
|
+
key
|
|
100
|
+
+ [
|
|
101
|
+
"__den__",
|
|
102
|
+
"__den_policy_clipped__",
|
|
103
|
+
"__den_instrument_overflow__",
|
|
104
|
+
"__den_bound_kind__",
|
|
105
|
+
]
|
|
106
|
+
]
|
|
107
|
+
|
|
108
|
+
# Join (lhs may be many-to-one vs rhs on the key)
|
|
109
|
+
merged = pd.merge(lhs, rhs, on=key, how="inner", validate="many_to_one")
|
|
110
|
+
|
|
111
|
+
# Only declared instrument lower bounds may explain a non-finite operand.
|
|
112
|
+
merged["__num__"] = pd.to_numeric(merged["__num__"], errors="coerce")
|
|
113
|
+
merged["__den__"] = pd.to_numeric(merged["__den__"], errors="coerce")
|
|
114
|
+
finite = np.isfinite(merged["__num__"]) & np.isfinite(merged["__den__"])
|
|
115
|
+
numerator_overflow = (
|
|
116
|
+
np.isposinf(merged["__num__"])
|
|
117
|
+
& merged["__num_instrument_overflow__"]
|
|
118
|
+
& merged["__num_bound_kind__"].eq("lower")
|
|
119
|
+
)
|
|
120
|
+
denominator_overflow = (
|
|
121
|
+
np.isposinf(merged["__den__"])
|
|
122
|
+
& merged["__den_instrument_overflow__"]
|
|
123
|
+
& merged["__den_bound_kind__"].eq("lower")
|
|
124
|
+
)
|
|
125
|
+
nonfinite = ~finite
|
|
126
|
+
omittable = (
|
|
127
|
+
nonfinite
|
|
128
|
+
& (np.isfinite(merged["__num__"]) | numerator_overflow)
|
|
129
|
+
& (np.isfinite(merged["__den__"]) | denominator_overflow)
|
|
130
|
+
)
|
|
131
|
+
unexpected_nonfinite = nonfinite & ~omittable
|
|
132
|
+
if unexpected_nonfinite.any():
|
|
133
|
+
raise ValueError("ratio: unexpected non-finite operand lacks instrument-overflow lower-bound provenance")
|
|
134
|
+
|
|
135
|
+
nonfinite_count = int(omittable.sum())
|
|
136
|
+
if nonfinite_count:
|
|
137
|
+
ctx.logger.warning(
|
|
138
|
+
"[warn]ratio[/warn] • %s: omitted %d aligned pair(s) with non-finite operands",
|
|
139
|
+
cfg.name,
|
|
140
|
+
nonfinite_count,
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
invalid = finite & (merged["__den__"] == 0)
|
|
144
|
+
dropped = int(invalid.sum())
|
|
145
|
+
if dropped:
|
|
146
|
+
ctx.logger.warning("[warn]ratio[/warn] • %s: dropped %d row(s) due to zero denominator", cfg.name, dropped)
|
|
147
|
+
|
|
148
|
+
merged = merged.loc[~omittable & ~invalid].copy()
|
|
149
|
+
bounded = merged["__num_bound_kind__"].ne("exact") | merged["__den_bound_kind__"].ne("exact")
|
|
150
|
+
nonpositive = merged["__num__"].le(0.0) | merged["__den__"].le(0.0)
|
|
151
|
+
if (bounded & nonpositive).any():
|
|
152
|
+
raise ValueError("ratio: bounded values require positive operands for directional bound propagation")
|
|
153
|
+
merged["value"] = merged["__num__"] / merged["__den__"]
|
|
154
|
+
merged["channel"] = cfg.name
|
|
155
|
+
merged["value_policy_clipped"] = merged["__num_policy_clipped__"] | merged["__den_policy_clipped__"]
|
|
156
|
+
merged["value_instrument_overflow"] = (
|
|
157
|
+
merged["__num_instrument_overflow__"] | merged["__den_instrument_overflow__"]
|
|
158
|
+
)
|
|
159
|
+
denominator_bounds = merged["__den_bound_kind__"].map(
|
|
160
|
+
{"exact": "exact", "lower": "upper", "upper": "lower", "indeterminate": "indeterminate"}
|
|
161
|
+
)
|
|
162
|
+
merged["value_bound_kind"] = [
|
|
163
|
+
_combine_bounds(numerator, denominator)
|
|
164
|
+
for numerator, denominator in zip(merged["__num_bound_kind__"], denominator_bounds, strict=True)
|
|
165
|
+
]
|
|
166
|
+
if emit_value_provenance and "overflow" in merged.columns:
|
|
167
|
+
merged["overflow"] = merged["value_instrument_overflow"]
|
|
168
|
+
|
|
169
|
+
# Restore original column set in original order (inherits metadata from lhs)
|
|
170
|
+
derived = merged[df.columns].copy()
|
|
171
|
+
|
|
172
|
+
out = pd.concat([df, derived], ignore_index=True)
|
|
173
|
+
if not emit_value_provenance:
|
|
174
|
+
# Generic ratios remain usable, but missing provenance is not evidence
|
|
175
|
+
# that an observation is exact. Four-state event-window ingestion rejects it.
|
|
176
|
+
out = out.loc[:, input_columns]
|
|
177
|
+
|
|
178
|
+
with suppress(Exception):
|
|
179
|
+
ctx.logger.info(
|
|
180
|
+
"ratio • [accent]%s[/accent] = %s / %s • +%d row(s) • keys=%s",
|
|
181
|
+
cfg.name,
|
|
182
|
+
cfg.numerator,
|
|
183
|
+
cfg.denominator,
|
|
184
|
+
len(derived),
|
|
185
|
+
key,
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
return {"df": out}
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _with_value_provenance(frame: pd.DataFrame) -> tuple[pd.DataFrame, bool]:
|
|
192
|
+
result = frame.copy()
|
|
193
|
+
explicit_fields = {"value_policy_clipped", "value_instrument_overflow", "value_bound_kind"}
|
|
194
|
+
present = explicit_fields & set(result.columns)
|
|
195
|
+
if present and present != explicit_fields:
|
|
196
|
+
raise ValueError("ratio: value provenance must provide all three explicit fields together")
|
|
197
|
+
if present:
|
|
198
|
+
policy_clipped = _strict_boolean(result["value_policy_clipped"], field="value_policy_clipped")
|
|
199
|
+
instrument_overflow = _strict_boolean(result["value_instrument_overflow"], field="value_instrument_overflow")
|
|
200
|
+
bounds = result["value_bound_kind"]
|
|
201
|
+
if bounds.isna().any() or not bounds.map(lambda value: isinstance(value, str)).all():
|
|
202
|
+
raise ValueError("ratio: value_bound_kind provenance must contain strings without missing values")
|
|
203
|
+
bounds = bounds.astype(str)
|
|
204
|
+
allowed = {"exact", "lower", "upper", "indeterminate"}
|
|
205
|
+
unknown = sorted(set(bounds) - allowed)
|
|
206
|
+
if unknown:
|
|
207
|
+
raise ValueError(f"ratio: unsupported value_bound_kind values: {unknown}")
|
|
208
|
+
affected = policy_clipped | instrument_overflow
|
|
209
|
+
if not affected.eq(bounds.ne("exact")).all():
|
|
210
|
+
raise ValueError("ratio: clipping and overflow provenance disagrees with value_bound_kind")
|
|
211
|
+
if "overflow" in result.columns:
|
|
212
|
+
observed_overflow = _strict_boolean(result["overflow"], field="overflow")
|
|
213
|
+
if not observed_overflow.eq(instrument_overflow).all():
|
|
214
|
+
raise ValueError("ratio: overflow disagrees with explicit instrument-overflow provenance")
|
|
215
|
+
else:
|
|
216
|
+
policy_clipped = pd.Series(False, index=result.index, dtype=bool)
|
|
217
|
+
instrument_overflow = pd.Series(False, index=result.index, dtype=bool)
|
|
218
|
+
bounds = pd.Series("exact", index=result.index, dtype=object)
|
|
219
|
+
result["value_policy_clipped"] = policy_clipped
|
|
220
|
+
result["value_instrument_overflow"] = instrument_overflow
|
|
221
|
+
result["value_bound_kind"] = bounds
|
|
222
|
+
return result, bool(present)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _strict_boolean(values: pd.Series, *, field: str) -> pd.Series:
|
|
226
|
+
if values.isna().any() or not values.map(lambda value: isinstance(value, (bool, np.bool_))).all():
|
|
227
|
+
raise ValueError(f"ratio: {field} provenance must contain booleans without missing values")
|
|
228
|
+
return values.astype(bool)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _combine_bounds(left: object, right: object) -> str:
|
|
232
|
+
bounds = {str(left), str(right)} - {"exact"}
|
|
233
|
+
if not bounds:
|
|
234
|
+
return "exact"
|
|
235
|
+
if len(bounds) == 1:
|
|
236
|
+
return bounds.pop()
|
|
237
|
+
return "indeterminate"
|