pisces-lite 3.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pisces_lite-3.0.0/LICENSE +21 -0
- pisces_lite-3.0.0/PKG-INFO +111 -0
- pisces_lite-3.0.0/README.md +76 -0
- pisces_lite-3.0.0/pyproject.toml +61 -0
- pisces_lite-3.0.0/setup.cfg +4 -0
- pisces_lite-3.0.0/src/pisces_lite/__init__.py +0 -0
- pisces_lite-3.0.0/src/pisces_lite/cv/__init__.py +15 -0
- pisces_lite-3.0.0/src/pisces_lite/cv/config.py +85 -0
- pisces_lite-3.0.0/src/pisces_lite/cv/fold.py +133 -0
- pisces_lite-3.0.0/src/pisces_lite/datasets/__init__.py +114 -0
- pisces_lite-3.0.0/src/pisces_lite/datasets/adapters.py +137 -0
- pisces_lite-3.0.0/src/pisces_lite/datasets/config.py +192 -0
- pisces_lite-3.0.0/src/pisces_lite/datasets/constants.py +188 -0
- pisces_lite-3.0.0/src/pisces_lite/datasets/data_set_object.py +323 -0
- pisces_lite-3.0.0/src/pisces_lite/datasets/id_extraction.py +185 -0
- pisces_lite-3.0.0/src/pisces_lite/datasets/loading.py +42 -0
- pisces_lite-3.0.0/src/pisces_lite/datasets/processing.py +238 -0
- pisces_lite-3.0.0/src/pisces_lite/metrics/__init__.py +45 -0
- pisces_lite-3.0.0/src/pisces_lite/metrics/config.py +112 -0
- pisces_lite-3.0.0/src/pisces_lite/metrics/context.py +31 -0
- pisces_lite-3.0.0/src/pisces_lite/metrics/etc.py +209 -0
- pisces_lite-3.0.0/src/pisces_lite/metrics/io.py +88 -0
- pisces_lite-3.0.0/src/pisces_lite/metrics/logger.py +172 -0
- pisces_lite-3.0.0/src/pisces_lite/metrics/specs/__init__.py +39 -0
- pisces_lite-3.0.0/src/pisces_lite/metrics/specs/auroc.py +53 -0
- pisces_lite-3.0.0/src/pisces_lite/metrics/specs/base.py +105 -0
- pisces_lite-3.0.0/src/pisces_lite/metrics/specs/builtin.py +165 -0
- pisces_lite-3.0.0/src/pisces_lite/model_io.py +139 -0
- pisces_lite-3.0.0/src/pisces_lite/plotting/__init__.py +6 -0
- pisces_lite-3.0.0/src/pisces_lite/proc/__init__.py +46 -0
- pisces_lite-3.0.0/src/pisces_lite/proc/_backend.py +32 -0
- pisces_lite-3.0.0/src/pisces_lite/proc/config.py +255 -0
- pisces_lite-3.0.0/src/pisces_lite/proc/constants.py +3 -0
- pisces_lite-3.0.0/src/pisces_lite/proc/features.py +64 -0
- pisces_lite-3.0.0/src/pisces_lite/proc/processing.py +853 -0
- pisces_lite-3.0.0/src/pisces_lite/roc/__init__.py +22 -0
- pisces_lite-3.0.0/src/pisces_lite/roc/roc_analysis.py +180 -0
- pisces_lite-3.0.0/src/pisces_lite/utils.py +71 -0
- pisces_lite-3.0.0/src/pisces_lite.egg-info/PKG-INFO +111 -0
- pisces_lite-3.0.0/src/pisces_lite.egg-info/SOURCES.txt +50 -0
- pisces_lite-3.0.0/src/pisces_lite.egg-info/dependency_links.txt +1 -0
- pisces_lite-3.0.0/src/pisces_lite.egg-info/requires.txt +16 -0
- pisces_lite-3.0.0/src/pisces_lite.egg-info/top_level.txt +1 -0
- pisces_lite-3.0.0/tests/test_cv.py +119 -0
- pisces_lite-3.0.0/tests/test_datasets.py +447 -0
- pisces_lite-3.0.0/tests/test_etc.py +125 -0
- pisces_lite-3.0.0/tests/test_id_extraction.py +60 -0
- pisces_lite-3.0.0/tests/test_metrics_logger.py +205 -0
- pisces_lite-3.0.0/tests/test_model_io.py +58 -0
- pisces_lite-3.0.0/tests/test_proc_config.py +59 -0
- pisces_lite-3.0.0/tests/test_roc.py +88 -0
- pisces_lite-3.0.0/tests/test_stacked_spectrograms.py +275 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Arcascope Inc
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pisces-lite
|
|
3
|
+
Version: 3.0.0
|
|
4
|
+
Summary: ML-backend-agnostic core supporting data set discovery, loading, common spectrogram/PSD-based processing, and CV scoring.
|
|
5
|
+
Author-email: Arcascope Inc <support@arcascope.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/Arcascope/pisces-lite
|
|
8
|
+
Project-URL: Repository, https://github.com/Arcascope/pisces-lite
|
|
9
|
+
Project-URL: Issues, https://github.com/Arcascope/pisces-lite/issues
|
|
10
|
+
Keywords: sleep,actigraphy,accelerometer,spectrogram,wearables
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering
|
|
18
|
+
Requires-Python: >=3.12
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: matplotlib
|
|
22
|
+
Requires-Dist: numpy
|
|
23
|
+
Requires-Dist: scipy
|
|
24
|
+
Requires-Dist: scikit-learn
|
|
25
|
+
Requires-Dist: pandas
|
|
26
|
+
Requires-Dist: seaborn
|
|
27
|
+
Requires-Dist: tqdm
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: pytest; extra == "dev"
|
|
30
|
+
Provides-Extra: proc
|
|
31
|
+
Requires-Dist: arcascope-senpy==4.0.3; extra == "proc"
|
|
32
|
+
Provides-Extra: jax
|
|
33
|
+
Requires-Dist: arcascope-senpy[jax]==4.0.3; extra == "jax"
|
|
34
|
+
Dynamic: license-file
|
|
35
|
+
|
|
36
|
+
# pisces-lite
|
|
37
|
+
|
|
38
|
+
ML-backend-agnostic core for wearable sleep data: dataset discovery and loading,
|
|
39
|
+
accelerometer-to-spectrogram processing, cross-validation splitting, and metrics
|
|
40
|
+
logging. It carries no model framework, though there is a JAX-accelerated option
|
|
41
|
+
for the `[jax]` extra components.
|
|
42
|
+
|
|
43
|
+
## Install
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
pip install pisces-lite
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Python 3.12+. To do the processing yourself rather than read a prebuilt feature
|
|
50
|
+
cache, install an extra — see [Optional extras](#optional-extras). The `proc`
|
|
51
|
+
and `jax` extras require a Python version the compiled backend publishes wheels
|
|
52
|
+
for (currently 3.12–3.14):
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
pip install "pisces-lite[proc]"
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## Getting started
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
from pisces_lite.datasets import DataSetObject, load_subject
|
|
62
|
+
|
|
63
|
+
# Discover datasets laid out as <dataset>/cleaned_<feature>/<subject>.csv
|
|
64
|
+
data_sets = DataSetObject.find_data_sets("data/")
|
|
65
|
+
ds = data_sets["mydata"]
|
|
66
|
+
|
|
67
|
+
# Load one subject: (accelerometer, PSG) frames with standardised columns
|
|
68
|
+
subject = load_subject(ds, ds.ids[0])
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Datasets are described by an optional `data_set.json` next to the data; it sets
|
|
72
|
+
per-feature timestamp units, accelerometer scaling, PSG stage mappings, CSV
|
|
73
|
+
delimiters, and subject-id patterns. Non-standard layouts can ship a
|
|
74
|
+
`pisces_lite_adapter.py` exposing `load_data_sets(root, **kwargs)`.
|
|
75
|
+
|
|
76
|
+
Configure the processing pipeline and turn raw accelerometer into spectrograms:
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
from pisces_lite.proc import ProcessingConfig
|
|
80
|
+
|
|
81
|
+
config = ProcessingConfig.from_dict({"type": "nufft", "fs": 32.0})
|
|
82
|
+
X = config.apply(accel_array) # (T, F) or (T, F, C) with spectral_channels
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Split subjects and score folds:
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
from pisces_lite.cv import CVConfig, iter_folds
|
|
89
|
+
from pisces_lite.metrics import MetricsConfig, MetricsLogger
|
|
90
|
+
|
|
91
|
+
cv = CVConfig(mode="loso", train_sets=["mydata"], test_sets=["mydata"])
|
|
92
|
+
for fold in iter_folds(cv, data_sets):
|
|
93
|
+
... # train on fold.train_subjects, evaluate fold.test_subjects
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## Optional extras
|
|
97
|
+
|
|
98
|
+
The base install never pulls a C++ toolchain. Add an extra only when you need
|
|
99
|
+
the processing pipeline itself.
|
|
100
|
+
|
|
101
|
+
| Extra | Install | What it adds | Use when |
|
|
102
|
+
|---|---|---|---|
|
|
103
|
+
| `proc` | `pip install "pisces-lite[proc]"` | `arcascope-senpy` (import name `senpy`; compiled pybind11 + finufft), the CPU `streaming`/`cpu` NUFFT backends | You are turning raw accelerometer data into spectrograms |
|
|
104
|
+
| `jax` | `pip install "pisces-lite[jax]"` | `arcascope-senpy[jax]` and JAX, adding the `jax` NUFFT backend | You want GPU-batched spectrogram extraction |
|
|
105
|
+
| `dev` | `pip install "pisces-lite[dev]"` | `pytest` | You are running the test suite |
|
|
106
|
+
|
|
107
|
+
Working from a prebuilt feature cache — a training or inference host that only
|
|
108
|
+
reads `cv`, `metrics`, `model_io`, and `ProcessingConfig` — needs no extra.
|
|
109
|
+
`pisces_lite.proc` resolves its pipeline classes lazily, so importing a
|
|
110
|
+
`ProcessingConfig` does not import `senpy`; touching a pipeline class without
|
|
111
|
+
the extra raises a clear error naming the extra to install.
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
# pisces-lite
|
|
2
|
+
|
|
3
|
+
ML-backend-agnostic core for wearable sleep data: dataset discovery and loading,
|
|
4
|
+
accelerometer-to-spectrogram processing, cross-validation splitting, and metrics
|
|
5
|
+
logging. It carries no model framework, though there is a JAX-accelerated option
|
|
6
|
+
for the `[jax]` extra components.
|
|
7
|
+
|
|
8
|
+
## Install
|
|
9
|
+
|
|
10
|
+
```bash
|
|
11
|
+
pip install pisces-lite
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Python 3.12+. To do the processing yourself rather than read a prebuilt feature
|
|
15
|
+
cache, install an extra — see [Optional extras](#optional-extras). The `proc`
|
|
16
|
+
and `jax` extras require a Python version the compiled backend publishes wheels
|
|
17
|
+
for (currently 3.12–3.14):
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install "pisces-lite[proc]"
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## Getting started
|
|
24
|
+
|
|
25
|
+
```python
|
|
26
|
+
from pisces_lite.datasets import DataSetObject, load_subject
|
|
27
|
+
|
|
28
|
+
# Discover datasets laid out as <dataset>/cleaned_<feature>/<subject>.csv
|
|
29
|
+
data_sets = DataSetObject.find_data_sets("data/")
|
|
30
|
+
ds = data_sets["mydata"]
|
|
31
|
+
|
|
32
|
+
# Load one subject: (accelerometer, PSG) frames with standardised columns
|
|
33
|
+
subject = load_subject(ds, ds.ids[0])
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Datasets are described by an optional `data_set.json` next to the data; it sets
|
|
37
|
+
per-feature timestamp units, accelerometer scaling, PSG stage mappings, CSV
|
|
38
|
+
delimiters, and subject-id patterns. Non-standard layouts can ship a
|
|
39
|
+
`pisces_lite_adapter.py` exposing `load_data_sets(root, **kwargs)`.
|
|
40
|
+
|
|
41
|
+
Configure the processing pipeline and turn raw accelerometer into spectrograms:
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from pisces_lite.proc import ProcessingConfig
|
|
45
|
+
|
|
46
|
+
config = ProcessingConfig.from_dict({"type": "nufft", "fs": 32.0})
|
|
47
|
+
X = config.apply(accel_array) # (T, F) or (T, F, C) with spectral_channels
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Split subjects and score folds:
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
from pisces_lite.cv import CVConfig, iter_folds
|
|
54
|
+
from pisces_lite.metrics import MetricsConfig, MetricsLogger
|
|
55
|
+
|
|
56
|
+
cv = CVConfig(mode="loso", train_sets=["mydata"], test_sets=["mydata"])
|
|
57
|
+
for fold in iter_folds(cv, data_sets):
|
|
58
|
+
... # train on fold.train_subjects, evaluate fold.test_subjects
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## Optional extras
|
|
62
|
+
|
|
63
|
+
The base install never pulls a C++ toolchain. Add an extra only when you need
|
|
64
|
+
the processing pipeline itself.
|
|
65
|
+
|
|
66
|
+
| Extra | Install | What it adds | Use when |
|
|
67
|
+
|---|---|---|---|
|
|
68
|
+
| `proc` | `pip install "pisces-lite[proc]"` | `arcascope-senpy` (import name `senpy`; compiled pybind11 + finufft), the CPU `streaming`/`cpu` NUFFT backends | You are turning raw accelerometer data into spectrograms |
|
|
69
|
+
| `jax` | `pip install "pisces-lite[jax]"` | `arcascope-senpy[jax]` and JAX, adding the `jax` NUFFT backend | You want GPU-batched spectrogram extraction |
|
|
70
|
+
| `dev` | `pip install "pisces-lite[dev]"` | `pytest` | You are running the test suite |
|
|
71
|
+
|
|
72
|
+
Working from a prebuilt feature cache — a training or inference host that only
|
|
73
|
+
reads `cv`, `metrics`, `model_io`, and `ProcessingConfig` — needs no extra.
|
|
74
|
+
`pisces_lite.proc` resolves its pipeline classes lazily, so importing a
|
|
75
|
+
`ProcessingConfig` does not import `senpy`; touching a pipeline class without
|
|
76
|
+
the extra raises a clear error naming the extra to install.
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pisces-lite"
|
|
7
|
+
version = "3.0.0"
|
|
8
|
+
description = "ML-backend-agnostic core supporting data set discovery, loading, common spectrogram/PSD-based processing, and CV scoring."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
authors = [{ name = "Arcascope Inc", email = "support@arcascope.com" }]
|
|
13
|
+
requires-python = ">=3.12"
|
|
14
|
+
keywords = ["sleep", "actigraphy", "accelerometer", "spectrogram", "wearables"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 4 - Beta",
|
|
17
|
+
"Intended Audience :: Science/Research",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3.12",
|
|
20
|
+
"Programming Language :: Python :: 3.13",
|
|
21
|
+
"Programming Language :: Python :: 3.14",
|
|
22
|
+
"Topic :: Scientific/Engineering",
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
# senpy is deliberately NOT here. It is a compiled extension (pybind11 +
|
|
26
|
+
# finufft) and is only ever used by pisces_lite.proc, which turns raw
|
|
27
|
+
# accelerometer data into spectrograms. A consumer working from an already
|
|
28
|
+
# built feature cache -- a training or inference host -- imports cv, metrics,
|
|
29
|
+
# model_io and ProcessingConfig but never calls into proc, so making it a hard
|
|
30
|
+
# dependency forced a C++ toolchain into images that had nothing to compile.
|
|
31
|
+
# Install "pisces-lite[proc]" to do the processing itself.
|
|
32
|
+
dependencies = [
|
|
33
|
+
"matplotlib",
|
|
34
|
+
"numpy",
|
|
35
|
+
"scipy",
|
|
36
|
+
"scikit-learn",
|
|
37
|
+
"pandas",
|
|
38
|
+
"seaborn",
|
|
39
|
+
"tqdm",
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
[project.optional-dependencies]
|
|
43
|
+
dev = ["pytest"]
|
|
44
|
+
# The processing pipeline: pisces_lite.proc's runtime backend.
|
|
45
|
+
# The distribution is named `arcascope-senpy`; its import name is `senpy`.
|
|
46
|
+
# (There is an unrelated `senpy` project on PyPI, hence the prefixed name.)
|
|
47
|
+
proc = [
|
|
48
|
+
"arcascope-senpy==4.0.3",
|
|
49
|
+
]
|
|
50
|
+
# proc on a JAX NUFFT backend. Superset of [proc]; arcascope-senpy[jax] covers it.
|
|
51
|
+
jax = [
|
|
52
|
+
"arcascope-senpy[jax]==4.0.3",
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
[project.urls]
|
|
56
|
+
Homepage = "https://github.com/Arcascope/pisces-lite"
|
|
57
|
+
Repository = "https://github.com/Arcascope/pisces-lite"
|
|
58
|
+
Issues = "https://github.com/Arcascope/pisces-lite/issues"
|
|
59
|
+
|
|
60
|
+
[tool.setuptools.packages.find]
|
|
61
|
+
where = ["src"]
|
|
File without changes
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Cross-validation orchestration (framework-agnostic).
|
|
2
|
+
|
|
3
|
+
Two modes:
|
|
4
|
+
|
|
5
|
+
- ``loso`` — leave-one-subject-out across the union of ``test_sets``.
|
|
6
|
+
- ``transfer`` — train once on ``train_sets``, evaluate every subject in
|
|
7
|
+
``test_sets`` (which must be disjoint from ``train_sets``).
|
|
8
|
+
|
|
9
|
+
Both modes hold out ``validation_sets`` from training. See
|
|
10
|
+
:class:`CVConfig` for the exact invariants.
|
|
11
|
+
"""
|
|
12
|
+
from pisces_lite.cv.config import CVConfig, CVMode
|
|
13
|
+
from pisces_lite.cv.fold import Fold, SubjectRef, iter_folds
|
|
14
|
+
|
|
15
|
+
__all__ = ["CVConfig", "CVMode", "Fold", "SubjectRef", "iter_folds"]
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""CVConfig: cross-validation splitting configuration (no ML-framework deps)."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import List, Literal, Optional
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
CVMode = Literal["loso", "transfer"]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass
|
|
14
|
+
class CVConfig:
|
|
15
|
+
"""Subject-splitting configuration.
|
|
16
|
+
|
|
17
|
+
Modes:
|
|
18
|
+
- ``loso`` (leave-one-subject-out): requires ``test_sets ⊆ train_sets``.
|
|
19
|
+
Iterates over every subject in the union of ``test_sets``, holding one
|
|
20
|
+
out per fold; trains on the remaining pooled train subjects (minus any
|
|
21
|
+
validation subjects). Produces N folds, N models.
|
|
22
|
+
- ``transfer``: requires ``train_sets ∩ test_sets = ∅``. Trains one
|
|
23
|
+
model on the pooled train subjects, evaluates every subject in the
|
|
24
|
+
union of ``test_sets``. Produces 1 fold, 1 model, |test| evaluations.
|
|
25
|
+
|
|
26
|
+
In both modes, ``train_sets ∩ validation_sets`` must be empty at the
|
|
27
|
+
dataset level; validation subjects are always excluded from training.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
mode: CVMode
|
|
31
|
+
train_sets: List[str]
|
|
32
|
+
test_sets: List[str]
|
|
33
|
+
validation_sets: List[str] = field(default_factory=list)
|
|
34
|
+
max_folds: Optional[int] = None
|
|
35
|
+
|
|
36
|
+
@classmethod
|
|
37
|
+
def from_json(cls, path: "Path | str") -> "CVConfig":
|
|
38
|
+
with open(path) as f:
|
|
39
|
+
return cls.from_dict(json.load(f))
|
|
40
|
+
|
|
41
|
+
@classmethod
|
|
42
|
+
def from_dict(cls, d: dict) -> "CVConfig":
|
|
43
|
+
known = {f for f in cls.__dataclass_fields__}
|
|
44
|
+
return cls(**{k: v for k, v in d.items() if k in known})
|
|
45
|
+
|
|
46
|
+
def validate(self) -> None:
|
|
47
|
+
"""Raise ValueError on any invariant violation."""
|
|
48
|
+
train = set(self.train_sets)
|
|
49
|
+
test = set(self.test_sets)
|
|
50
|
+
val = set(self.validation_sets)
|
|
51
|
+
|
|
52
|
+
if not train:
|
|
53
|
+
raise ValueError("CVConfig: train_sets is empty")
|
|
54
|
+
if not test:
|
|
55
|
+
raise ValueError("CVConfig: test_sets is empty")
|
|
56
|
+
|
|
57
|
+
overlap = train & val
|
|
58
|
+
if overlap:
|
|
59
|
+
raise ValueError(
|
|
60
|
+
f"CVConfig: train_sets and validation_sets overlap on "
|
|
61
|
+
f"{sorted(overlap)}; validation must be a held-out dataset"
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
if self.mode == "loso":
|
|
65
|
+
extras = test - train
|
|
66
|
+
if extras:
|
|
67
|
+
raise ValueError(
|
|
68
|
+
f"CVConfig(mode='loso'): requires test_sets ⊆ train_sets; "
|
|
69
|
+
f"found in test_sets but not train_sets: {sorted(extras)}. "
|
|
70
|
+
f"If you intended to train on {sorted(train)} and test on "
|
|
71
|
+
f"disjoint datasets, use mode='transfer'."
|
|
72
|
+
)
|
|
73
|
+
elif self.mode == "transfer":
|
|
74
|
+
overlap = train & test
|
|
75
|
+
if overlap:
|
|
76
|
+
raise ValueError(
|
|
77
|
+
f"CVConfig(mode='transfer'): requires train_sets ∩ "
|
|
78
|
+
f"test_sets = ∅; overlap: {sorted(overlap)}. "
|
|
79
|
+
f"If you intended leave-one-subject-out within that "
|
|
80
|
+
f"overlap, use mode='loso'."
|
|
81
|
+
)
|
|
82
|
+
else:
|
|
83
|
+
raise ValueError(
|
|
84
|
+
f"CVConfig: mode must be 'loso' or 'transfer', got {self.mode!r}"
|
|
85
|
+
)
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""Fold iteration over datasets according to a CVConfig."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from typing import Dict, Iterable, Iterator, List, NamedTuple, Sequence, Union
|
|
6
|
+
|
|
7
|
+
from pisces_lite.cv.config import CVConfig
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class SubjectRef(NamedTuple):
|
|
11
|
+
"""A reference to one subject. Cheap; consumers pass through ``load_subject``."""
|
|
12
|
+
|
|
13
|
+
dataset_name: str
|
|
14
|
+
subject_id: str
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True)
|
|
18
|
+
class Fold:
|
|
19
|
+
"""One fold of a CV run.
|
|
20
|
+
|
|
21
|
+
For ``mode='transfer'``, always ``fold_id=0`` and ``total_folds=1``.
|
|
22
|
+
For ``mode='loso'``, ``fold_id`` ranges ``0..total_folds-1`` and each fold
|
|
23
|
+
holds out exactly one ``test_subjects`` entry.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
fold_id: int
|
|
27
|
+
total_folds: int
|
|
28
|
+
mode: str
|
|
29
|
+
description: str
|
|
30
|
+
train_subjects: List[SubjectRef]
|
|
31
|
+
val_subjects: List[SubjectRef]
|
|
32
|
+
test_subjects: List[SubjectRef]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
DatasetsArg = Union[Dict[str, "object"], Sequence["object"]]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _datasets_by_name(datasets: DatasetsArg) -> Dict[str, "object"]:
|
|
39
|
+
if isinstance(datasets, dict):
|
|
40
|
+
return datasets
|
|
41
|
+
return {d.name: d for d in datasets}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _subjects_from(
|
|
45
|
+
set_names: Iterable[str],
|
|
46
|
+
by_name: Dict[str, "object"],
|
|
47
|
+
exclude: Iterable[SubjectRef] = (),
|
|
48
|
+
) -> List[SubjectRef]:
|
|
49
|
+
excluded = set(exclude)
|
|
50
|
+
out: List[SubjectRef] = []
|
|
51
|
+
for name in set_names:
|
|
52
|
+
ds = by_name[name]
|
|
53
|
+
for sid in ds.ids:
|
|
54
|
+
ref = SubjectRef(name, sid)
|
|
55
|
+
if ref in excluded:
|
|
56
|
+
continue
|
|
57
|
+
out.append(ref)
|
|
58
|
+
return sorted(out)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _require_all_loaded(cv_config: CVConfig, by_name: Dict[str, "object"]) -> None:
|
|
62
|
+
needed = set(cv_config.train_sets) | set(cv_config.test_sets) | set(cv_config.validation_sets)
|
|
63
|
+
missing = needed - set(by_name)
|
|
64
|
+
if missing:
|
|
65
|
+
raise KeyError(
|
|
66
|
+
f"iter_folds: datasets referenced by CVConfig are not loaded: "
|
|
67
|
+
f"{sorted(missing)}. Loaded: {sorted(by_name)}"
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def iter_folds(cv_config: CVConfig, datasets: DatasetsArg) -> Iterator[Fold]:
|
|
72
|
+
"""Yield one :class:`Fold` per CV split, dispatching on ``cv_config.mode``.
|
|
73
|
+
|
|
74
|
+
``datasets`` may be a list of ``DataSetObject`` or a pre-built ``{name: ds}``
|
|
75
|
+
mapping. The caller is responsible for calling ``parse_data`` on each
|
|
76
|
+
dataset before passing it here.
|
|
77
|
+
"""
|
|
78
|
+
cv_config.validate()
|
|
79
|
+
by_name = _datasets_by_name(datasets)
|
|
80
|
+
_require_all_loaded(cv_config, by_name)
|
|
81
|
+
|
|
82
|
+
val_subjects = _subjects_from(cv_config.validation_sets, by_name)
|
|
83
|
+
|
|
84
|
+
if cv_config.mode == "transfer":
|
|
85
|
+
train_subjects = _subjects_from(
|
|
86
|
+
cv_config.train_sets, by_name, exclude=val_subjects
|
|
87
|
+
)
|
|
88
|
+
test_subjects = _subjects_from(cv_config.test_sets, by_name)
|
|
89
|
+
yield Fold(
|
|
90
|
+
fold_id=0,
|
|
91
|
+
total_folds=1,
|
|
92
|
+
mode="transfer",
|
|
93
|
+
description=(
|
|
94
|
+
f"transfer: 1 model on {len(train_subjects)} train subj "
|
|
95
|
+
f"from {list(cv_config.train_sets)} → "
|
|
96
|
+
f"{len(test_subjects)} test subj "
|
|
97
|
+
f"from {list(cv_config.test_sets)} "
|
|
98
|
+
f"(val: {len(val_subjects)})"
|
|
99
|
+
),
|
|
100
|
+
train_subjects=train_subjects,
|
|
101
|
+
val_subjects=val_subjects,
|
|
102
|
+
test_subjects=test_subjects,
|
|
103
|
+
)
|
|
104
|
+
return
|
|
105
|
+
|
|
106
|
+
# LOSO
|
|
107
|
+
test_pool = _subjects_from(
|
|
108
|
+
cv_config.test_sets, by_name, exclude=val_subjects
|
|
109
|
+
)
|
|
110
|
+
train_pool = _subjects_from(
|
|
111
|
+
cv_config.train_sets, by_name, exclude=val_subjects
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
total = len(test_pool)
|
|
115
|
+
if cv_config.max_folds is not None:
|
|
116
|
+
test_pool = test_pool[: cv_config.max_folds]
|
|
117
|
+
|
|
118
|
+
for i, held_out in enumerate(test_pool):
|
|
119
|
+
train_subjects = [s for s in train_pool if s != held_out]
|
|
120
|
+
yield Fold(
|
|
121
|
+
fold_id=i,
|
|
122
|
+
total_folds=total,
|
|
123
|
+
mode="loso",
|
|
124
|
+
description=(
|
|
125
|
+
f"LOSO fold {i + 1}/{total}: holding out "
|
|
126
|
+
f"{held_out.subject_id} from {held_out.dataset_name} "
|
|
127
|
+
f"({len(train_subjects)} train subj, "
|
|
128
|
+
f"{len(val_subjects)} val subj)"
|
|
129
|
+
),
|
|
130
|
+
train_subjects=train_subjects,
|
|
131
|
+
val_subjects=val_subjects,
|
|
132
|
+
test_subjects=[held_out],
|
|
133
|
+
)
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""Multi-feature sleep-study dataset loading.
|
|
2
|
+
|
|
3
|
+
Datasets are described by a per-dataset :class:`DataSetConfig` parsed from
|
|
4
|
+
``data_set.json`` rather than sniffed from the dataset name.
|
|
5
|
+
"""
|
|
6
|
+
from pisces_lite.datasets.constants import (
|
|
7
|
+
ACC_HZ,
|
|
8
|
+
DELTA_T_COL,
|
|
9
|
+
END_HZ,
|
|
10
|
+
MINIMUM_ACCEL_SAMPLES_PER_PSG,
|
|
11
|
+
MINUTES_TO_SECONDS,
|
|
12
|
+
PSG_CLASS_TO_NAME,
|
|
13
|
+
PSG_COL,
|
|
14
|
+
PSG_DT,
|
|
15
|
+
PSG_MAPPING_5C,
|
|
16
|
+
PSG_MAPPING_DREAMT,
|
|
17
|
+
PSG_MAPPING_NO_N4,
|
|
18
|
+
PSG_MAPPING_PRESETS,
|
|
19
|
+
PSG_MAPPING_WEAVER,
|
|
20
|
+
PSG_5C_MAPPING_TO_WLDR,
|
|
21
|
+
PSG_5C_MAPPING_TO_WNR,
|
|
22
|
+
PSG_5C_MAPPING_TO_WS,
|
|
23
|
+
PSG_MASK,
|
|
24
|
+
PSG_STAGING_LEGEND,
|
|
25
|
+
SET_NAMES_PRETTY,
|
|
26
|
+
SLEEP_CLASS_LABEL,
|
|
27
|
+
TIMESTAMP_COL,
|
|
28
|
+
WAKE_CLASS_LABEL,
|
|
29
|
+
WALCH_PSG_LABELS,
|
|
30
|
+
X_COL,
|
|
31
|
+
Y_COL,
|
|
32
|
+
Z_COL,
|
|
33
|
+
get_class_names,
|
|
34
|
+
)
|
|
35
|
+
from pisces_lite.datasets.config import (
|
|
36
|
+
AccelConfig,
|
|
37
|
+
CSVConfig,
|
|
38
|
+
DataSetConfig,
|
|
39
|
+
PSGConfig,
|
|
40
|
+
SubjectData,
|
|
41
|
+
TimestampConfig,
|
|
42
|
+
load_subject,
|
|
43
|
+
)
|
|
44
|
+
from pisces_lite.datasets.adapters import (
|
|
45
|
+
DEFAULT_ADAPTER_FILENAME,
|
|
46
|
+
DataSetAdapter,
|
|
47
|
+
import_adapter_module,
|
|
48
|
+
load_data_sets_from_adapter,
|
|
49
|
+
normalize_data_sets,
|
|
50
|
+
)
|
|
51
|
+
from pisces_lite.datasets.data_set_object import DataSetObject, get_subject_data
|
|
52
|
+
from pisces_lite.datasets.id_extraction import IdExtractor, SimplifiablePrefixTree
|
|
53
|
+
from pisces_lite.datasets.loading import determine_header_rows_and_delimiter
|
|
54
|
+
from pisces_lite.datasets.processing import (
|
|
55
|
+
align_trim,
|
|
56
|
+
calculate_binary_activity,
|
|
57
|
+
mask_data,
|
|
58
|
+
psg_map,
|
|
59
|
+
psg_to_sleep_wake,
|
|
60
|
+
rescale_spectrogram,
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
__all__ = [
|
|
64
|
+
"ACC_HZ",
|
|
65
|
+
"AccelConfig",
|
|
66
|
+
"CSVConfig",
|
|
67
|
+
"DEFAULT_ADAPTER_FILENAME",
|
|
68
|
+
"DELTA_T_COL",
|
|
69
|
+
"DataSetAdapter",
|
|
70
|
+
"DataSetConfig",
|
|
71
|
+
"DataSetObject",
|
|
72
|
+
"PSGConfig",
|
|
73
|
+
"SubjectData",
|
|
74
|
+
"TimestampConfig",
|
|
75
|
+
"load_subject",
|
|
76
|
+
"END_HZ",
|
|
77
|
+
"IdExtractor",
|
|
78
|
+
"MINIMUM_ACCEL_SAMPLES_PER_PSG",
|
|
79
|
+
"MINUTES_TO_SECONDS",
|
|
80
|
+
"PSG_CLASS_TO_NAME",
|
|
81
|
+
"PSG_COL",
|
|
82
|
+
"PSG_DT",
|
|
83
|
+
"PSG_MAPPING_5C",
|
|
84
|
+
"PSG_MAPPING_DREAMT",
|
|
85
|
+
"PSG_MAPPING_NO_N4",
|
|
86
|
+
"PSG_MAPPING_PRESETS",
|
|
87
|
+
"PSG_MAPPING_WEAVER",
|
|
88
|
+
"PSG_5C_MAPPING_TO_WLDR",
|
|
89
|
+
"PSG_5C_MAPPING_TO_WNR",
|
|
90
|
+
"PSG_5C_MAPPING_TO_WS",
|
|
91
|
+
"PSG_MASK",
|
|
92
|
+
"PSG_STAGING_LEGEND",
|
|
93
|
+
"SET_NAMES_PRETTY",
|
|
94
|
+
"SLEEP_CLASS_LABEL",
|
|
95
|
+
"SimplifiablePrefixTree",
|
|
96
|
+
"TIMESTAMP_COL",
|
|
97
|
+
"WAKE_CLASS_LABEL",
|
|
98
|
+
"WALCH_PSG_LABELS",
|
|
99
|
+
"X_COL",
|
|
100
|
+
"Y_COL",
|
|
101
|
+
"Z_COL",
|
|
102
|
+
"align_trim",
|
|
103
|
+
"calculate_binary_activity",
|
|
104
|
+
"determine_header_rows_and_delimiter",
|
|
105
|
+
"get_class_names",
|
|
106
|
+
"get_subject_data",
|
|
107
|
+
"import_adapter_module",
|
|
108
|
+
"load_data_sets_from_adapter",
|
|
109
|
+
"mask_data",
|
|
110
|
+
"normalize_data_sets",
|
|
111
|
+
"psg_map",
|
|
112
|
+
"psg_to_sleep_wake",
|
|
113
|
+
"rescale_spectrogram",
|
|
114
|
+
]
|