electropycal 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- electropycal/__init__.py +7 -0
- electropycal/_demo.py +145 -0
- electropycal/_parallel.py +40 -0
- electropycal/analysis_config.py +168 -0
- electropycal/cli.py +323 -0
- electropycal/data/__init__.py +1 -0
- electropycal/data/inventory.py +479 -0
- electropycal/data/io.py +68 -0
- electropycal/data/pstrace.py +205 -0
- electropycal/data/quality.py +74 -0
- electropycal/data/schema.py +118 -0
- electropycal/data/stabilization.py +234 -0
- electropycal/data/synthetic.py +520 -0
- electropycal/deployment/__init__.py +1 -0
- electropycal/deployment/deploy.py +244 -0
- electropycal/deployment/domain.py +88 -0
- electropycal/deployment/plots.py +45 -0
- electropycal/diagnostics/__init__.py +1 -0
- electropycal/diagnostics/variance.py +217 -0
- electropycal/diagreview.py +770 -0
- electropycal/discovery/__init__.py +1 -0
- electropycal/discovery/baseline.py +55 -0
- electropycal/discovery/batch.py +91 -0
- electropycal/discovery/config.py +205 -0
- electropycal/discovery/review.py +157 -0
- electropycal/discovery/runner.py +362 -0
- electropycal/discovery/scheduler.py +243 -0
- electropycal/evaluation/__init__.py +1 -0
- electropycal/evaluation/admissibility.py +85 -0
- electropycal/evaluation/baselines.py +181 -0
- electropycal/evaluation/classify.py +88 -0
- electropycal/evaluation/cv.py +140 -0
- electropycal/evaluation/framing.py +210 -0
- electropycal/evaluation/hierarchical.py +161 -0
- electropycal/evaluation/metrics.py +173 -0
- electropycal/evaluation/multioutput.py +130 -0
- electropycal/evaluation/stratify.py +121 -0
- electropycal/evaluation/tracks.py +35 -0
- electropycal/features/__init__.py +1 -0
- electropycal/features/catalog.py +115 -0
- electropycal/features/eis.py +135 -0
- electropycal/features/extract.py +951 -0
- electropycal/features/fscv.py +448 -0
- electropycal/features/normalize.py +168 -0
- electropycal/features/pin.py +267 -0
- electropycal/features/targets.py +193 -0
- electropycal/models/__init__.py +1 -0
- electropycal/models/base.py +66 -0
- electropycal/models/plsr.py +141 -0
- electropycal/models/variants.py +217 -0
- electropycal/overview.py +144 -0
- electropycal/py.typed +0 -0
- electropycal/qcdash.py +467 -0
- electropycal/rawspectra.py +824 -0
- electropycal/selection/__init__.py +1 -0
- electropycal/selection/cars.py +102 -0
- electropycal/selection/icc.py +32 -0
- electropycal/selection/pseudo_multivariate.py +82 -0
- electropycal/selection/univariate.py +23 -0
- electropycal/stabreview.py +301 -0
- electropycal/viz.py +97 -0
- electropycal-0.9.0.dist-info/METADATA +156 -0
- electropycal-0.9.0.dist-info/RECORD +66 -0
- electropycal-0.9.0.dist-info/WHEEL +4 -0
- electropycal-0.9.0.dist-info/entry_points.txt +2 -0
- electropycal-0.9.0.dist-info/licenses/LICENSE +21 -0
electropycal/__init__.py
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""electropycal — data processing and recalibration for electrochemical sensors."""
|
|
2
|
+
__version__ = "0.9.0"
|
|
3
|
+
|
|
4
|
+
from .features.catalog import (FEATURE_DEFINITIONS, catalog_frame, feature_catalog,
|
|
5
|
+
print_feature_catalog)
|
|
6
|
+
|
|
7
|
+
__all__ = ["FEATURE_DEFINITIONS", "catalog_frame", "feature_catalog", "print_feature_catalog"]
|
electropycal/_demo.py
ADDED
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
"""Locate the shipped demo tree, from wherever a notebook happens to be running.
|
|
2
|
+
|
|
3
|
+
**Private on purpose.** The leading underscore keeps this out of the 0.9.0 public API
|
|
4
|
+
(same convention as :mod:`electropycal._parallel`): it is scaffolding so the shipped
|
|
5
|
+
notebooks have something to open, not a path API anyone should build on. It lives
|
|
6
|
+
*inside the package* rather than next to the notebooks because in Colab the notebook is
|
|
7
|
+
uploaded on its own and the pip-installed wheel is the only thing present — a sibling
|
|
8
|
+
``notebooks/_demo.py`` would not be importable there.
|
|
9
|
+
|
|
10
|
+
The problem it solves: every public notebook used to hardcode ``ROOT =
|
|
11
|
+
"demo/in_vitro/input"``, a path relative to the repo root. A Jupyter kernel's working
|
|
12
|
+
directory is the *notebook's own* directory, so opening ``notebooks/x.ipynb`` and
|
|
13
|
+
running it gave ``FileNotFoundError: 'demo\\in_vitro\\input'`` several cells in. Swapping
|
|
14
|
+
in ``"../demo/..."`` just moves the breakage: it would then work from ``notebooks/`` and
|
|
15
|
+
fail from the repo root, from VS Code with a workspace-root cwd, and in Colab.
|
|
16
|
+
|
|
17
|
+
So nothing here counts parent directories. Candidate roots are probed in order and each
|
|
18
|
+
is accepted only if the tree it should contain is really there:
|
|
19
|
+
|
|
20
|
+
1. ``$ELECTROPYCAL_DEMO`` — explicit override, for CI or an unpacked release.
|
|
21
|
+
2. the current working directory and each of its ancestors — covers the repo root,
|
|
22
|
+
``notebooks/``, ``publish/``, and anywhere else inside a checkout.
|
|
23
|
+
3. the installed package's own location — covers an editable/src-layout install driven
|
|
24
|
+
from an unrelated cwd. A wheel install has no demo tree next to it, so this simply
|
|
25
|
+
does not match, which is the correct answer rather than a wrong guess.
|
|
26
|
+
|
|
27
|
+
If none match — the ordinary Colab case, where there is no checkout at all —
|
|
28
|
+
:func:`demo_input` synthesizes an equivalent tree instead of failing.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import os
|
|
34
|
+
import tempfile
|
|
35
|
+
from pathlib import Path
|
|
36
|
+
|
|
37
|
+
__all__ = ["DemoTreeNotFound", "demo_input", "find_demo_tree", "require_dir"]
|
|
38
|
+
|
|
39
|
+
#: Demo kinds and the generator that can stand in for each when nothing is on disk.
|
|
40
|
+
_KINDS = {"in_vitro": "write_synthetic_pstrace_dir", "in_vivo": "write_synthetic_invivo_dir"}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class DemoTreeNotFound(FileNotFoundError):
|
|
44
|
+
"""Raised when the demo tree cannot be located and synthesis was not wanted."""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _candidate_roots() -> list[Path]:
|
|
48
|
+
"""Roots that might contain a ``demo/`` directory, most explicit first."""
|
|
49
|
+
roots: list[Path] = []
|
|
50
|
+
|
|
51
|
+
env = os.environ.get("ELECTROPYCAL_DEMO")
|
|
52
|
+
if env:
|
|
53
|
+
p = Path(env).expanduser()
|
|
54
|
+
# Accept either the tree root (containing demo/) or demo/ itself.
|
|
55
|
+
roots += [p, p.parent]
|
|
56
|
+
|
|
57
|
+
cwd = Path.cwd().resolve()
|
|
58
|
+
roots += [cwd, *cwd.parents]
|
|
59
|
+
|
|
60
|
+
# src-layout editable install: .../<root>/src/electropycal/_demo.py -> parents[2] is <root>.
|
|
61
|
+
pkg = Path(__file__).resolve()
|
|
62
|
+
roots += [pkg.parents[2], pkg.parents[1]] if len(pkg.parents) > 2 else []
|
|
63
|
+
|
|
64
|
+
seen, uniq = set(), []
|
|
65
|
+
for r in roots:
|
|
66
|
+
if r not in seen:
|
|
67
|
+
seen.add(r)
|
|
68
|
+
uniq.append(r)
|
|
69
|
+
return uniq
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _relative_targets(kind: str) -> list[Path]:
|
|
73
|
+
"""Sub-paths, relative to a candidate root, that would BE the demo input tree."""
|
|
74
|
+
return [Path("demo") / kind / "input", Path(kind) / "input"]
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def find_demo_tree(kind: str = "in_vitro") -> Path | None:
|
|
78
|
+
"""Return the shipped demo *input* tree for ``kind``, or ``None`` if unreachable.
|
|
79
|
+
|
|
80
|
+
``None`` is a normal outcome, not an error: it means no checkout is present (Colab),
|
|
81
|
+
and the caller should synthesize a tree instead.
|
|
82
|
+
"""
|
|
83
|
+
if kind not in _KINDS:
|
|
84
|
+
raise ValueError(f"unknown demo kind {kind!r}; expected one of {sorted(_KINDS)}")
|
|
85
|
+
for root in _candidate_roots():
|
|
86
|
+
for rel in _relative_targets(kind):
|
|
87
|
+
cand = root / rel
|
|
88
|
+
if cand.is_dir() and any(cand.iterdir()):
|
|
89
|
+
return cand.resolve()
|
|
90
|
+
return None
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _searched_report(kind: str) -> str:
|
|
94
|
+
"""Human-readable account of what was looked for and where — never a bare errno."""
|
|
95
|
+
rels = " or ".join(str(r) for r in _relative_targets(kind))
|
|
96
|
+
lines = [f"could not find the shipped demo tree for kind={kind!r}.",
|
|
97
|
+
f" looked for : {rels}",
|
|
98
|
+
f" under : {len(_candidate_roots())} candidate root(s):"]
|
|
99
|
+
lines += [f" {r}" for r in _candidate_roots()[:12]]
|
|
100
|
+
lines += ["",
|
|
101
|
+
" Fixes, in order of least surprise:",
|
|
102
|
+
" - run scripts/build_demo_dataset.py --out demo (builds it, ~15 s)",
|
|
103
|
+
" - set ELECTROPYCAL_DEMO=/path/to/the/tree/containing/demo",
|
|
104
|
+
" - point ROOT at your own PSTrace export directory instead"]
|
|
105
|
+
return "\n".join(lines)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def require_dir(path, what: str = "directory") -> str:
|
|
109
|
+
"""Validate a caller-supplied path, reporting what was expected rather than an errno."""
|
|
110
|
+
p = Path(path).expanduser()
|
|
111
|
+
if p.is_dir():
|
|
112
|
+
return str(p.resolve())
|
|
113
|
+
raise DemoTreeNotFound(
|
|
114
|
+
f"{what} not found: {p!r}\n"
|
|
115
|
+
f" resolved to : {p.resolve() if not p.is_absolute() else p}\n"
|
|
116
|
+
f" cwd : {Path.cwd()}\n"
|
|
117
|
+
f" A relative path is resolved against the KERNEL's working directory, which for a\n"
|
|
118
|
+
f" notebook is the notebook's own folder -- not the repository root.")
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def demo_input(kind: str = "in_vitro", user_path=None, *,
|
|
122
|
+
synthesize: bool = True, **synth_kwargs) -> str:
|
|
123
|
+
"""One call that every public notebook can use for its input tree.
|
|
124
|
+
|
|
125
|
+
``user_path`` -- if given, validated and returned (clear error if absent).
|
|
126
|
+
otherwise -- the shipped demo tree if any candidate root has one,
|
|
127
|
+
otherwise -- a freshly synthesized equivalent, unless ``synthesize=False``.
|
|
128
|
+
|
|
129
|
+
Returns a string path, because the review classes take ``str | Path`` and notebooks
|
|
130
|
+
print it.
|
|
131
|
+
"""
|
|
132
|
+
if user_path is not None:
|
|
133
|
+
return require_dir(user_path, f"{kind} input directory")
|
|
134
|
+
|
|
135
|
+
found = find_demo_tree(kind)
|
|
136
|
+
if found is not None:
|
|
137
|
+
return str(found)
|
|
138
|
+
|
|
139
|
+
if not synthesize:
|
|
140
|
+
raise DemoTreeNotFound(_searched_report(kind))
|
|
141
|
+
|
|
142
|
+
from . import data # noqa: F401 (ensure the subpackage is importable before getattr)
|
|
143
|
+
from .data import synthetic
|
|
144
|
+
writer = getattr(synthetic, _KINDS[kind])
|
|
145
|
+
return str(writer(tempfile.mkdtemp(prefix=f"electropycal-demo-{kind}-"), **synth_kwargs))
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Shared joblib helper for the library's parallel work units.
|
|
2
|
+
|
|
3
|
+
Two callers: the session-parallel raw walks (extraction, QC, reliability), one unit per
|
|
4
|
+
device-timepoint session, and discovery's fold loop, one unit per CV fold. Both are
|
|
5
|
+
embarrassingly parallel, so they parallelize cleanly. The one hazard for **non-notebook / direct-API** callers is
|
|
6
|
+
BLAS/OpenMP oversubscription: without the thread pin the notebooks set, each worker process would
|
|
7
|
+
spin up a full multi-threaded BLAS pool, so ``n_jobs`` workers × ``cores`` threads fight over the
|
|
8
|
+
cores and the "speedup" can be a slowdown. :func:`map_sessions` closes that hole by capping every
|
|
9
|
+
worker to a single inner thread (``inner_max_num_threads=1``) itself — so a plain
|
|
10
|
+
``extract_dataset(n_jobs=8)`` from a script is safe with no environment setup required.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from collections.abc import Iterable
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def should_parallelize(n_jobs: int, n_items: int) -> bool:
|
|
20
|
+
"""True when a joblib pool is worth spawning: more than one worker requested (``n_jobs`` not
|
|
21
|
+
0/1; negatives like -1 mean "all cores") and more than one item to spread over it. Below that,
|
|
22
|
+
the worker spawn + pickling overhead outweighs the gain, so the caller runs serially."""
|
|
23
|
+
return n_jobs not in (0, 1) and n_items > 1
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def map_sessions(n_jobs: int, tasks: Iterable[Any]) -> list:
|
|
27
|
+
"""Run ``tasks`` (an iterable of ``joblib.delayed(...)`` calls) across worker processes, with
|
|
28
|
+
**each worker capped to one BLAS/OpenMP thread** so parallel sessions never oversubscribe the
|
|
29
|
+
cores — even when the caller has not pinned BLAS in its environment. joblib preserves input
|
|
30
|
+
order, so the returned list is in ``tasks`` order (the raw walks rely on this for deterministic,
|
|
31
|
+
serial-identical output). The caller decides *whether* to parallelize (see
|
|
32
|
+
:func:`should_parallelize`); this only runs the pool once that decision is made.
|
|
33
|
+
|
|
34
|
+
Named for its first caller; ``tasks`` are any independent work units — raw-walk sessions or
|
|
35
|
+
discovery CV folds."""
|
|
36
|
+
from joblib import Parallel, parallel_config
|
|
37
|
+
# inner_max_num_threads=1: the library defends itself against oversubscription regardless of the
|
|
38
|
+
# caller's environment (notebook pin, script with nothing set, CI). loky = process backend.
|
|
39
|
+
with parallel_config(backend="loky", inner_max_num_threads=1):
|
|
40
|
+
return list(Parallel(n_jobs=n_jobs)(tasks))
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""Shared analysis configuration for the review/QC/discovery notebooks.
|
|
2
|
+
|
|
3
|
+
The three pre-made notebooks — ``raw_spectra_review``, ``quality_filtering_dashboard``,
|
|
4
|
+
and ``discovery_checkpointed`` — run as **separate** Colab kernels, so a config set in
|
|
5
|
+
one cannot be seen by the others. To keep the EIS ``band`` and the QC / extraction
|
|
6
|
+
parameters from drifting apart between them, this module persists a single
|
|
7
|
+
``electropycal_analysis_config.json`` **inside the data ``ROOT``**. Each notebook loads it
|
|
8
|
+
(falling back to defaults if absent) and may still override any field in its setup cell.
|
|
9
|
+
|
|
10
|
+
Workflow:
|
|
11
|
+
1. In ``raw_spectra_review`` set the band + params and call ``save_analysis_config(ROOT, cfg)``.
|
|
12
|
+
2. ``quality_filtering_dashboard`` and ``discovery_checkpointed`` call
|
|
13
|
+
``load_analysis_config(ROOT)`` and get exactly those values.
|
|
14
|
+
|
|
15
|
+
The library primitives (``extract_dataset``, ``channel_quality_report``, the CLI) still take
|
|
16
|
+
explicit parameters — this is a convenience layer for the notebooks, not a new gate.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import json
|
|
22
|
+
from dataclasses import asdict, dataclass, field
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
CONFIG_FILENAME = "electropycal_analysis_config.json"
|
|
26
|
+
QC_STATS_FILENAME = "electropycal_qc_stats.json"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class AnalysisConfig:
|
|
31
|
+
"""Analysis **parameters** shared by the raw-spectra / QC-dashboard / discovery notebooks.
|
|
32
|
+
|
|
33
|
+
This holds only processing/QC knobs that must agree across notebooks (band, peak method,
|
|
34
|
+
acceptance, gates). It deliberately does **not** hold data *selection* (device types / ids /
|
|
35
|
+
channels / timepoints): selection is per-notebook-run (raw_spectra_review is often run one
|
|
36
|
+
device at a time, while discovery runs across many), so each notebook keeps its own selection
|
|
37
|
+
and only the parameters are shared.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
band: tuple[float, float] | str = (2.0, 2000.0) # (lo, hi) Hz; "auto" for data-driven. Study
|
|
41
|
+
#: default 2–2000 Hz: the f-dependent drift-alignment review (diagnostics §3.1) shows aligned
|
|
42
|
+
#: drift concentrated ~2 Hz–2 kHz, while retention is flat below the ~10 kHz inductive-onset cliff,
|
|
43
|
+
#: so 2 kHz captures the signal at full retention and 2 Hz reaches the low-f aligned band.
|
|
44
|
+
peak_method: str = "direct" # "direct" | "chord"
|
|
45
|
+
acceptance: str = "monotonic" # extract_dataset acceptance test
|
|
46
|
+
mono_tol: float = 0.10 # EIS.2 |Z|-monotonicity tolerance
|
|
47
|
+
min_norm_snr: float = 3.0 # per-dose reproducibility-SNR cutoff
|
|
48
|
+
monotonic_r_min: float = 0.6 # dose-monotonicity min log-conc correlation
|
|
49
|
+
mono_method: str = "pearson" # dose-response corr: "pearson" | "spearman" (rank)
|
|
50
|
+
max_reps: int | None = 3 # first N FSCV replicate cycles
|
|
51
|
+
gate_on: dict = field(default_factory=lambda: {
|
|
52
|
+
"eis": True, "monotonic": True, "snr_all": False, "peak_in_window": False})
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def config_path(root: str | Path) -> Path:
|
|
56
|
+
"""Where the shared config lives for a given data ``root``."""
|
|
57
|
+
return Path(root) / CONFIG_FILENAME
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def load_analysis_config(root: str | Path, **overrides) -> AnalysisConfig:
|
|
61
|
+
"""Load the shared config from ``<root>/electropycal_analysis_config.json``.
|
|
62
|
+
|
|
63
|
+
Returns defaults when the file is absent. Any keyword in ``overrides`` that is not
|
|
64
|
+
``None`` replaces the loaded value (so a notebook cell can override a field inline).
|
|
65
|
+
Unknown keys in the file are ignored, so old configs keep loading as fields are added.
|
|
66
|
+
"""
|
|
67
|
+
data: dict = {}
|
|
68
|
+
p = config_path(root)
|
|
69
|
+
if p.exists():
|
|
70
|
+
data = json.loads(p.read_text())
|
|
71
|
+
known = {f for f in AnalysisConfig().__dataclass_fields__}
|
|
72
|
+
cfg = AnalysisConfig(**{k: v for k, v in data.items() if k in known})
|
|
73
|
+
if isinstance(cfg.band, list): # JSON has no tuples
|
|
74
|
+
cfg.band = tuple(cfg.band)
|
|
75
|
+
for k, v in overrides.items():
|
|
76
|
+
if v is not None and k in known:
|
|
77
|
+
setattr(cfg, k, v)
|
|
78
|
+
return cfg
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def save_analysis_config(root: str | Path, cfg: AnalysisConfig) -> Path:
|
|
82
|
+
"""Write ``cfg`` to ``<root>/electropycal_analysis_config.json`` and return the path."""
|
|
83
|
+
p = config_path(root)
|
|
84
|
+
p.write_text(json.dumps(asdict(cfg), indent=2, default=list))
|
|
85
|
+
return p
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def qc_stats_path(root: str | Path) -> Path:
|
|
89
|
+
"""Where the QC-stats companion artifact lives for a given data ``root``."""
|
|
90
|
+
return Path(root) / QC_STATS_FILENAME
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def save_qc_stats(root: str | Path, stats: dict) -> Path:
|
|
94
|
+
"""Write the QC-statistics companion to ``<root>/electropycal_qc_stats.json``.
|
|
95
|
+
|
|
96
|
+
This is the *evidence* alongside the config (``save_analysis_config``): the config records
|
|
97
|
+
**what** QC parameters were used, this records **what they did** on the data (yields, per-gate
|
|
98
|
+
dropout, the stringency-sweep numbers) so a finalized run documents both the knobs and their
|
|
99
|
+
effect. ``stats`` should embed the config that produced it (e.g. under a ``"config"`` key).
|
|
100
|
+
Non-JSON scalars (numpy types) are coerced via ``default=str``.
|
|
101
|
+
"""
|
|
102
|
+
p = qc_stats_path(root)
|
|
103
|
+
p.write_text(json.dumps(stats, indent=2, default=str))
|
|
104
|
+
return p
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def load_qc_stats(root: str | Path) -> dict | None:
|
|
108
|
+
"""Load the QC-stats companion, or ``None`` if it has not been written yet."""
|
|
109
|
+
p = qc_stats_path(root)
|
|
110
|
+
return json.loads(p.read_text()) if p.exists() else None
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def format_provenance(root: str | Path, target: str | None = None,
|
|
114
|
+
cfg: AnalysisConfig | None = None) -> str:
|
|
115
|
+
"""One human-readable block: the pre-processing **parameters**, the **decisions**, and (if the
|
|
116
|
+
QC-stats companion exists) their **impact** on the dataset. Printed at the start of every discovery
|
|
117
|
+
run — whatever the entry point (notebook / script / CLI) — so a run always states what
|
|
118
|
+
pre-processing produced the featureset it trained on, and why."""
|
|
119
|
+
cfg = cfg or load_analysis_config(root)
|
|
120
|
+
stats = load_qc_stats(root)
|
|
121
|
+
b = cfg.band if isinstance(cfg.band, str) else f"({cfg.band[0]:g}, {cfg.band[1]:g}) Hz"
|
|
122
|
+
L = ["=" * 74,
|
|
123
|
+
"ElectroPyCal — run provenance (pre-processing parameters · decisions · impact)",
|
|
124
|
+
"=" * 74,
|
|
125
|
+
"config (electropycal_analysis_config.json):",
|
|
126
|
+
f" band = {b} peak_method = {cfg.peak_method} acceptance = {cfg.acceptance}",
|
|
127
|
+
f" EIS.2 mono_tol = {cfg.mono_tol} FSCV.1 monotonic_r_min = {cfg.monotonic_r_min} "
|
|
128
|
+
f"({cfg.mono_method}) min_norm_snr = {cfg.min_norm_snr} max_reps = {cfg.max_reps}",
|
|
129
|
+
f" gate_on = {cfg.gate_on}",
|
|
130
|
+
"decisions / conventions:",
|
|
131
|
+
f" target = {target or 'NormIpeak'}",
|
|
132
|
+
" D0-normalization: ON (per-sensor drift-from-baseline; additive for phase/bounded types, "
|
|
133
|
+
"multiplicative for magnitudes)",
|
|
134
|
+
" leakage-safe predictors: EIS + 0 nM-background FSCV only; faradaic peak features are "
|
|
135
|
+
"RESERVED targets",
|
|
136
|
+
" NormIpeak < 0 dose rows dropped (drop_negative, non-physical)"]
|
|
137
|
+
if stats:
|
|
138
|
+
t = stats.get("totals", {})
|
|
139
|
+
L.append("QC impact (electropycal_qc_stats.json):")
|
|
140
|
+
L.append(f" paired = {t.get('channel_timepoints_paired')} retained = "
|
|
141
|
+
f"{t.get('overall_valid')} ({t.get('overall_valid_pct')}%)")
|
|
142
|
+
for g in (stats.get("gate_impact") or []):
|
|
143
|
+
L.append(f" {g.get('gate')}: {g.get('n_channeltimepoints_failed')} failed "
|
|
144
|
+
f"({g.get('failed_%')}%)")
|
|
145
|
+
else:
|
|
146
|
+
L.append("QC impact: electropycal_qc_stats.json not found "
|
|
147
|
+
"(run quality_filtering_dashboard §7 to generate it)")
|
|
148
|
+
L += ["see docs/DESIGN.md for the method rationale.", "=" * 74]
|
|
149
|
+
return "\n".join(L)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def print_provenance(root: str | Path, target: str | None = None,
|
|
153
|
+
cfg: AnalysisConfig | None = None) -> None:
|
|
154
|
+
"""Print :func:`format_provenance` (safe: never raises — provenance is informational)."""
|
|
155
|
+
try:
|
|
156
|
+
print(format_provenance(root, target=target, cfg=cfg), flush=True)
|
|
157
|
+
except Exception as e: # never let provenance break a run
|
|
158
|
+
print(f"(provenance unavailable: {e})", flush=True)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def resolve_band(root: str | Path, cfg: AnalysisConfig) -> tuple[float, float]:
|
|
162
|
+
"""Resolve ``cfg.band`` to a concrete ``(lo, hi)`` — ``"auto"`` → ``recommended_band(root)``."""
|
|
163
|
+
if isinstance(cfg.band, str):
|
|
164
|
+
if cfg.band != "auto":
|
|
165
|
+
raise ValueError(f"band must be a (lo, hi) tuple or 'auto', got {cfg.band!r}")
|
|
166
|
+
from .data.inventory import recommended_band
|
|
167
|
+
return recommended_band(root)
|
|
168
|
+
return (float(cfg.band[0]), float(cfg.band[1]))
|