electrotrace 1.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- electrotrace-1.9.0/LICENSE +20 -0
- electrotrace-1.9.0/PKG-INFO +187 -0
- electrotrace-1.9.0/README.md +132 -0
- electrotrace-1.9.0/pyproject.toml +94 -0
- electrotrace-1.9.0/setup.cfg +4 -0
- electrotrace-1.9.0/src/electrotrace/__init__.py +28 -0
- electrotrace-1.9.0/src/electrotrace/__main__.py +3 -0
- electrotrace-1.9.0/src/electrotrace/annotations.py +194 -0
- electrotrace-1.9.0/src/electrotrace/baseline_detectors.py +122 -0
- electrotrace-1.9.0/src/electrotrace/beats.py +47 -0
- electrotrace-1.9.0/src/electrotrace/benchmark.py +154 -0
- electrotrace-1.9.0/src/electrotrace/candidate_suppressor.py +441 -0
- electrotrace-1.9.0/src/electrotrace/cli.py +630 -0
- electrotrace-1.9.0/src/electrotrace/detectors.py +152 -0
- electrotrace-1.9.0/src/electrotrace/formats.py +147 -0
- electrotrace-1.9.0/src/electrotrace/fp_analysis.py +358 -0
- electrotrace-1.9.0/src/electrotrace/hearttwin_adapter.py +90 -0
- electrotrace-1.9.0/src/electrotrace/io.py +163 -0
- electrotrace-1.9.0/src/electrotrace/lead_quality.py +138 -0
- electrotrace-1.9.0/src/electrotrace/lead_selection.py +169 -0
- electrotrace-1.9.0/src/electrotrace/metadata.py +111 -0
- electrotrace-1.9.0/src/electrotrace/ml.py +178 -0
- electrotrace-1.9.0/src/electrotrace/phenotype.py +84 -0
- electrotrace-1.9.0/src/electrotrace/phenotype_validation.py +133 -0
- electrotrace-1.9.0/src/electrotrace/polarity_v2.py +201 -0
- electrotrace-1.9.0/src/electrotrace/project.py +43 -0
- electrotrace-1.9.0/src/electrotrace/project_store.py +147 -0
- electrotrace-1.9.0/src/electrotrace/provenance.py +155 -0
- electrotrace-1.9.0/src/electrotrace/qrs_delineation.py +139 -0
- electrotrace-1.9.0/src/electrotrace/qtdb_detector_adapter.py +7 -0
- electrotrace-1.9.0/src/electrotrace/research_validation.py +146 -0
- electrotrace-1.9.0/src/electrotrace/scale_estimation.py +172 -0
- electrotrace-1.9.0/src/electrotrace/security.py +57 -0
- electrotrace-1.9.0/src/electrotrace/server_app.py +375 -0
- electrotrace-1.9.0/src/electrotrace/signal.py +84 -0
- electrotrace-1.9.0/src/electrotrace/statistics.py +98 -0
- electrotrace-1.9.0/src/electrotrace/threshold_selection.py +40 -0
- electrotrace-1.9.0/src/electrotrace/validation.py +241 -0
- electrotrace-1.9.0/src/electrotrace/validation_detectors.py +570 -0
- electrotrace-1.9.0/src/electrotrace/wfdb_records.py +375 -0
- electrotrace-1.9.0/src/electrotrace/window.py +117 -0
- electrotrace-1.9.0/src/electrotrace.egg-info/PKG-INFO +187 -0
- electrotrace-1.9.0/src/electrotrace.egg-info/SOURCES.txt +106 -0
- electrotrace-1.9.0/src/electrotrace.egg-info/dependency_links.txt +1 -0
- electrotrace-1.9.0/src/electrotrace.egg-info/entry_points.txt +3 -0
- electrotrace-1.9.0/src/electrotrace.egg-info/requires.txt +38 -0
- electrotrace-1.9.0/src/electrotrace.egg-info/top_level.txt +1 -0
- electrotrace-1.9.0/tests/test_annotations.py +54 -0
- electrotrace-1.9.0/tests/test_api.py +96 -0
- electrotrace-1.9.0/tests/test_audit_calibration.py +18 -0
- electrotrace-1.9.0/tests/test_audit_hardening.py +38 -0
- electrotrace-1.9.0/tests/test_audit_regressions.py +65 -0
- electrotrace-1.9.0/tests/test_baseline_detectors.py +21 -0
- electrotrace-1.9.0/tests/test_beats_ml.py +64 -0
- electrotrace-1.9.0/tests/test_candidate_suppressor.py +75 -0
- electrotrace-1.9.0/tests/test_certified_wfdb_mitdb_locked.py +131 -0
- electrotrace-1.9.0/tests/test_derive_polarity_thresholds_mitdb_extended.py +308 -0
- electrotrace-1.9.0/tests/test_detector_registry.py +12 -0
- electrotrace-1.9.0/tests/test_diagnose_merge_incart.py +34 -0
- electrotrace-1.9.0/tests/test_dual_polarity_merge.py +267 -0
- electrotrace-1.9.0/tests/test_edb_lead_selector_development.py +72 -0
- electrotrace-1.9.0/tests/test_edb_posthoc_failures.py +140 -0
- electrotrace-1.9.0/tests/test_edb_prospective.py +241 -0
- electrotrace-1.9.0/tests/test_evaluate_v2_gate.py +427 -0
- electrotrace-1.9.0/tests/test_fp_analysis.py +149 -0
- electrotrace-1.9.0/tests/test_hardening.py +44 -0
- electrotrace-1.9.0/tests/test_hearttwin_adapter.py +23 -0
- electrotrace-1.9.0/tests/test_incart_complete_characterization.py +144 -0
- electrotrace-1.9.0/tests/test_incart_forensics_scripts.py +187 -0
- electrotrace-1.9.0/tests/test_io.py +31 -0
- electrotrace-1.9.0/tests/test_lead_quality.py +139 -0
- electrotrace-1.9.0/tests/test_lead_selection.py +90 -0
- electrotrace-1.9.0/tests/test_lead_selection_v3.py +40 -0
- electrotrace-1.9.0/tests/test_ltafdb_label_free_features.py +87 -0
- electrotrace-1.9.0/tests/test_ltafdb_lead_selector_prospective.py +224 -0
- electrotrace-1.9.0/tests/test_ltafdb_posthoc_selector_audit.py +62 -0
- electrotrace-1.9.0/tests/test_ltstdb_zymed_selector_v3_posthoc.py +137 -0
- electrotrace-1.9.0/tests/test_ltstdb_zymed_selector_v3_prospective.py +124 -0
- electrotrace-1.9.0/tests/test_metadata.py +32 -0
- electrotrace-1.9.0/tests/test_model_persistence.py +55 -0
- electrotrace-1.9.0/tests/test_phenotype_validation.py +32 -0
- electrotrace-1.9.0/tests/test_polarity_v2.py +72 -0
- electrotrace-1.9.0/tests/test_polarity_width_override.py +104 -0
- electrotrace-1.9.0/tests/test_probability_alignment_bugfix.py +95 -0
- electrotrace-1.9.0/tests/test_project_api.py +21 -0
- electrotrace-1.9.0/tests/test_project_store_window.py +28 -0
- electrotrace-1.9.0/tests/test_provenance_and_research_validation.py +135 -0
- electrotrace-1.9.0/tests/test_public_api.py +28 -0
- electrotrace-1.9.0/tests/test_qrs_delineation.py +39 -0
- electrotrace-1.9.0/tests/test_qtdb_tolerance_curve.py +135 -0
- electrotrace-1.9.0/tests/test_recalibration_and_inspection.py +538 -0
- electrotrace-1.9.0/tests/test_record_calibration.py +26 -0
- electrotrace-1.9.0/tests/test_recovery.py +36 -0
- electrotrace-1.9.0/tests/test_release_metadata.py +47 -0
- electrotrace-1.9.0/tests/test_research.py +61 -0
- electrotrace-1.9.0/tests/test_security_and_project_store.py +40 -0
- electrotrace-1.9.0/tests/test_signal.py +27 -0
- electrotrace-1.9.0/tests/test_stage1_scale_estimator.py +174 -0
- electrotrace-1.9.0/tests/test_stage2_feature_scale.py +77 -0
- electrotrace-1.9.0/tests/test_svdb_selector_v2_posthoc.py +75 -0
- electrotrace-1.9.0/tests/test_svdb_selector_v2_prospective.py +298 -0
- electrotrace-1.9.0/tests/test_two_stage_detector.py +63 -0
- electrotrace-1.9.0/tests/test_v2_gate_confidence.py +152 -0
- electrotrace-1.9.0/tests/test_validation.py +65 -0
- electrotrace-1.9.0/tests/test_validation_detector.py +19 -0
- electrotrace-1.9.0/tests/test_validation_detectors.py +35 -0
- electrotrace-1.9.0/tests/test_wfdb_records.py +238 -0
- electrotrace-1.9.0/tests/test_window.py +33 -0
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
GNU AFFERO GENERAL PUBLIC LICENSE
|
|
2
|
+
Version 3, 19 November 2007
|
|
3
|
+
|
|
4
|
+
Copyright (c) 2026 Virelion Biotech
|
|
5
|
+
|
|
6
|
+
This program is free software: you can redistribute it and/or modify
|
|
7
|
+
it under the terms of the GNU Affero General Public License as published by
|
|
8
|
+
the Free Software Foundation, either version 3 of the License, or
|
|
9
|
+
(at your option) any later version.
|
|
10
|
+
|
|
11
|
+
This program is distributed in the hope that it will be useful,
|
|
12
|
+
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
13
|
+
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
14
|
+
GNU Affero General Public License for more details.
|
|
15
|
+
|
|
16
|
+
You should have received a copy of the GNU Affero General Public License
|
|
17
|
+
along with this program. If not, see <https://www.gnu.org/licenses/>.
|
|
18
|
+
|
|
19
|
+
For the full license text, see:
|
|
20
|
+
https://www.gnu.org/licenses/agpl-3.0.html
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: electrotrace
|
|
3
|
+
Version: 1.9.0
|
|
4
|
+
Summary: Reproducible ECG/electrophysiology annotation, benchmarking, phenotyping, and leakage-safe validation toolkit
|
|
5
|
+
License: AGPL-3.0-or-later
|
|
6
|
+
Project-URL: Homepage, https://github.com/Virelion-Biotech/Virelion-ElectroTrace
|
|
7
|
+
Project-URL: Documentation, https://github.com/Virelion-Biotech/Virelion-ElectroTrace/tree/main/docs
|
|
8
|
+
Project-URL: Repository, https://github.com/Virelion-Biotech/Virelion-ElectroTrace
|
|
9
|
+
Project-URL: Issues, https://github.com/Virelion-Biotech/Virelion-ElectroTrace/issues
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: License :: OSI Approved :: GNU Affero General Public License v3 or later (AGPLv3+)
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering
|
|
20
|
+
Requires-Python: >=3.10
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
License-File: LICENSE
|
|
23
|
+
Requires-Dist: numpy>=1.24
|
|
24
|
+
Requires-Dist: pandas>=2.0
|
|
25
|
+
Requires-Dist: scipy>=1.10
|
|
26
|
+
Requires-Dist: scikit-learn>=1.4
|
|
27
|
+
Provides-Extra: wfdb
|
|
28
|
+
Requires-Dist: wfdb>=4.1; extra == "wfdb"
|
|
29
|
+
Provides-Extra: edf
|
|
30
|
+
Requires-Dist: pyedflib>=0.1.38; extra == "edf"
|
|
31
|
+
Provides-Extra: server
|
|
32
|
+
Requires-Dist: flask>=3.0; extra == "server"
|
|
33
|
+
Provides-Extra: models
|
|
34
|
+
Requires-Dist: skops>=0.14; extra == "models"
|
|
35
|
+
Provides-Extra: all
|
|
36
|
+
Requires-Dist: wfdb>=4.1; extra == "all"
|
|
37
|
+
Requires-Dist: pyedflib>=0.1.38; extra == "all"
|
|
38
|
+
Requires-Dist: flask>=3.0; extra == "all"
|
|
39
|
+
Requires-Dist: skops>=0.14; extra == "all"
|
|
40
|
+
Provides-Extra: test
|
|
41
|
+
Requires-Dist: pytest>=8.0; extra == "test"
|
|
42
|
+
Requires-Dist: wfdb>=4.1; extra == "test"
|
|
43
|
+
Requires-Dist: pyedflib>=0.1.38; extra == "test"
|
|
44
|
+
Requires-Dist: skops>=0.14; extra == "test"
|
|
45
|
+
Provides-Extra: dev
|
|
46
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
47
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
48
|
+
Requires-Dist: build>=1.2; extra == "dev"
|
|
49
|
+
Requires-Dist: mypy>=1.11; extra == "dev"
|
|
50
|
+
Requires-Dist: wfdb>=4.1; extra == "dev"
|
|
51
|
+
Requires-Dist: pyedflib>=0.1.38; extra == "dev"
|
|
52
|
+
Requires-Dist: flask>=3.0; extra == "dev"
|
|
53
|
+
Requires-Dist: skops>=0.14; extra == "dev"
|
|
54
|
+
Dynamic: license-file
|
|
55
|
+
|
|
56
|
+
# ElectroTrace
|
|
57
|
+
|
|
58
|
+
**Current source version:** 1.9.0 (release candidate)
|
|
59
|
+
|
|
60
|
+
**A reproducible ECG/electrophysiology annotation and benchmarking workbench.**
|
|
61
|
+
|
|
62
|
+
ElectroTrace is built around a simple principle: **the evidence is the product**. It provides one place to run detectors, compare them under declared protocols, preserve record/subject-level statistics, and carry hashes, software versions, and provenance alongside results.
|
|
63
|
+
|
|
64
|
+
The repository also ships its own Stage-1 and two-stage Random Forest detector, but that detector is not the project's only or primary claim. The current evidence is deliberately mixed: the two-stage model performs strongly on its locked MIT-BIH split, and two successive model generations have both shown a cross-database drop on INCART. That gap is preserved and documented rather than hidden.
|
|
65
|
+
|
|
66
|
+
## Current evidence
|
|
67
|
+
|
|
68
|
+
**Two distinct model generations exist for the two-stage detector and must not be conflated** (see `validation_reports/VALIDATION_STATUS.md` for the full history). The original locked MIT-BIH row below is the `candidate-features-v3` model from 1.8.1. The current `candidate-features-v4` / windowed-std model now also has explicit MIT-BIH legacy non-regression audits, including the leakage-safe 0.0/0.0 polarity-gate result (F1 0.9644). Those audits do not become clean prospective evidence because MIT-BIH historically informed the adaptive-polarity mechanism.
|
|
69
|
+
|
|
70
|
+
| Protocol | Detector | Model generation | Records | Sensitivity | PPV | F1 |
|
|
71
|
+
|---|---|---|---:|---:|---:|---:|
|
|
72
|
+
| Locked MIT-BIH held-out | ElectroTrace two-stage | v3 (1.8.1 lock) | 12 | 0.9924 | 0.9879 | 0.9902 |
|
|
73
|
+
| MIT-BIH leakage-safe gate audit | ElectroTrace two-stage | v4 (legacy non-regression) | 12 | 0.9390 | 0.9913 | 0.9644 |
|
|
74
|
+
| Locked MIT-BIH held-out | Pan-Tompkins reimplementation | — | 12 | 0.9908 | 0.9954 | 0.9931 |
|
|
75
|
+
| Locked MIT-BIH held-out | Hamilton reimplementation | — | 12 | 0.9990 | 0.9303 | 0.9634 |
|
|
76
|
+
| Locked MIT-BIH held-out | ElectroTrace Stage-1 | — | 12 | 0.9931 | 0.7553 | 0.8580 |
|
|
77
|
+
| INCART external comparison | ElectroTrace two-stage | v3 | 68 | 0.3239 | 0.9679 | 0.4854 |
|
|
78
|
+
| INCART external comparison | ElectroTrace two-stage | v4 (windowed-std) | 68 | 0.9011 | 0.7594 | 0.8242 |
|
|
79
|
+
| INCART certified WFDB comparison | WFDB gqrs | — | 68 | 0.9324 | 0.9265 | 0.9294 |
|
|
80
|
+
| INCART complete source cohort | ElectroTrace two-stage | v4 frozen / development data | 75 | 0.8977 | 0.7633 | 0.8251 |
|
|
81
|
+
| INCART complete source cohort | WFDB gqrs | certified reference baseline | 75 | 0.9372 | 0.9310 | 0.9341 |
|
|
82
|
+
| INCART complete source cohort | WFDB sqrs | certified reference baseline | 75 | 0.7617 | 0.9537 | 0.8469 |
|
|
83
|
+
| European ST-T prospective external | ElectroTrace two-stage | v4 frozen | 90 | 0.9056 | 0.9644 | 0.9341 |
|
|
84
|
+
|
|
85
|
+
These values come from the repository's locked/archived validation artifacts. They are research results, not clinical validation and not evidence of universal detector superiority. **INCART informed v4 feature-scaling development and remains exposed development data**, so neither its historical 68-record row nor the completed 75-record row is a clean generalization estimate. The 75-record completion exists to finish the source cohort reproducibly, including the seven formerly malformed edge annotations under a two-certified-detector repair rule. A later one-shot prospective evaluation on all 90 European ST-T Database records provides independent external evidence for the unchanged frozen v4 configuration (F1 0.9341), but does not make INCART held-out or validate future post-hoc retuning. See `validation_reports/VALIDATION_STATUS.md` and `docs/CROSS_DATABASE_POLICY.md` for the evidence boundaries.
|
|
86
|
+
|
|
87
|
+
## Why ElectroTrace exists
|
|
88
|
+
|
|
89
|
+
Most ECG software answers "what detector should I use?" ElectroTrace is aimed at a different question:
|
|
90
|
+
|
|
91
|
+
> **Under one explicit protocol, how does this detector behave, and can someone else reproduce the answer?**
|
|
92
|
+
|
|
93
|
+
Primary statistics should treat records or subjects as the experimental unit rather than pretending individual beats are independent biological replicates. Validation artifacts can carry dataset hashes, detector configuration, software version, git commit, and protocol metadata.
|
|
94
|
+
|
|
95
|
+
### Primary use cases
|
|
96
|
+
|
|
97
|
+
**Preclinical electrophysiology:** batch animal recordings, retain explicit subject identifiers, and export record/subject-level tables for downstream statistics.
|
|
98
|
+
|
|
99
|
+
**Methods-heavy human ECG research:** evaluate an existing detector on held-out records and retain an auditable artifact for papers, reviews, and lab handoff.
|
|
100
|
+
|
|
101
|
+
**Detector development:** register a detector plugin and compare it against fixed baselines under the same matching rules.
|
|
102
|
+
|
|
103
|
+
ElectroTrace is research software. It is not a clinical device and the current models are not validated for clinical deployment.
|
|
104
|
+
|
|
105
|
+
## Install
|
|
106
|
+
|
|
107
|
+
Python 3.10+ is required.
|
|
108
|
+
|
|
109
|
+
**v1.9.0 is release-ready, but this repository does not claim PyPI availability until the tagged Trusted Publishing workflow succeeds.** Until then, install from source:
|
|
110
|
+
|
|
111
|
+
~~~bash
|
|
112
|
+
git clone https://github.com/Virelion-Biotech/Virelion-ElectroTrace.git
|
|
113
|
+
cd Virelion-ElectroTrace
|
|
114
|
+
python -m pip install -e ".[test,dev]"
|
|
115
|
+
~~~
|
|
116
|
+
|
|
117
|
+
## Five-minute start
|
|
118
|
+
|
|
119
|
+
~~~bash
|
|
120
|
+
electrotrace list
|
|
121
|
+
electrotrace detect recording.edf --detector pan-tompkins --channel 0 -o peaks.csv
|
|
122
|
+
electrotrace batch data/ --detector pan-tompkins --workers 4 -o results/
|
|
123
|
+
electrotrace report validation.json -o validation.html
|
|
124
|
+
~~~
|
|
125
|
+
|
|
126
|
+
Batch outputs:
|
|
127
|
+
|
|
128
|
+
~~~text
|
|
129
|
+
results/
|
|
130
|
+
├── beats.csv
|
|
131
|
+
├── records.csv
|
|
132
|
+
├── subjects.csv
|
|
133
|
+
├── failures.csv
|
|
134
|
+
├── manifest.json
|
|
135
|
+
├── batch_state.json
|
|
136
|
+
└── peaks/
|
|
137
|
+
~~~
|
|
138
|
+
|
|
139
|
+
Subject information is never inferred from filenames. Supply a two-column CSV with record and subject_id using --subject-map when subject aggregation is appropriate.
|
|
140
|
+
|
|
141
|
+
## Validation and benchmarking
|
|
142
|
+
|
|
143
|
+
~~~bash
|
|
144
|
+
electrotrace validate .cache/physionet/mitdb/100 --detector pan-tompkins -o validation.json
|
|
145
|
+
electrotrace bench .cache/physionet/mitdb --detectors pan-tompkins,hamilton --tolerance-ms 75 -o bench.json
|
|
146
|
+
~~~
|
|
147
|
+
|
|
148
|
+
Bench currently operates on local WFDB records. The repository does not vendor PhysioNet data.
|
|
149
|
+
|
|
150
|
+
The frozen experimental QRS boundary locator has also completed a 105-record QT
|
|
151
|
+
Database tolerance-curve study. Reference-centered joint onset/offset success
|
|
152
|
+
was 0.589 at 20 ms, 0.845 at 40 ms, 0.925 at 60 ms, 0.954 at 80 ms, and 0.994
|
|
153
|
+
at 100 ms, with record-level bootstrap uncertainty. This is delineation
|
|
154
|
+
characterization, not an end-to-end detector or clinical-performance claim;
|
|
155
|
+
see `docs/QTDB_VALIDATION.md` and `docs/QRS_DELINEATION.md`.
|
|
156
|
+
|
|
157
|
+
The detector plugin interface is already present. The next benchmark phase will add external package adapters and a regenerated multi-database leaderboard rather than a single scalar ranking.
|
|
158
|
+
|
|
159
|
+
## Model artifacts
|
|
160
|
+
|
|
161
|
+
The two-stage Random Forest is opt-in. New model persistence uses .skops plus a metadata sidecar. Legacy pickle models are migration-only and rejected by default.
|
|
162
|
+
|
|
163
|
+
A model records its feature schema and training scikit-learn major version. Loading stops when the runtime major version is incompatible or when the .skops file contains unknown serialized types.
|
|
164
|
+
|
|
165
|
+
The complete 75-record INCART result is part of the evidence surface: cross-database behavior must be measured instead of inferred from MIT-BIH. The research-use policy for unseen domains is documented in `docs/CROSS_DATABASE_POLICY.md`; ElectroTrace does not silently infer that a new database is in-domain or automatically promote the RF path over established reference baselines.
|
|
166
|
+
|
|
167
|
+
## Validation philosophy
|
|
168
|
+
|
|
169
|
+
- record-level rather than beat-level primary statistics;
|
|
170
|
+
- one-to-one matching under a declared tolerance;
|
|
171
|
+
- locked splits with explicit seeds;
|
|
172
|
+
- bootstrap confidence intervals over records;
|
|
173
|
+
- SHA-256 input hashes and software/git provenance;
|
|
174
|
+
- separate timing error, sensitivity, PPV, and F1;
|
|
175
|
+
- explicit limitations and non-claims.
|
|
176
|
+
|
|
177
|
+
## What is not claimed
|
|
178
|
+
|
|
179
|
+
ElectroTrace does not claim clinical performance, real-time streaming performance, population generalization, or superiority over established QRS detectors.
|
|
180
|
+
|
|
181
|
+
## Documentation and citation
|
|
182
|
+
|
|
183
|
+
See docs/, mkdocs.yml, and CITATION.cff. The repository includes Zenodo metadata for v1.9.0, but a DOI is not claimed until an actual tagged release is archived by Zenodo.
|
|
184
|
+
|
|
185
|
+
## License
|
|
186
|
+
|
|
187
|
+
AGPL-3.0-or-later at present. Any future core/server license split will be an explicit project decision.
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
# ElectroTrace
|
|
2
|
+
|
|
3
|
+
**Current source version:** 1.9.0 (release candidate)
|
|
4
|
+
|
|
5
|
+
**A reproducible ECG/electrophysiology annotation and benchmarking workbench.**
|
|
6
|
+
|
|
7
|
+
ElectroTrace is built around a simple principle: **the evidence is the product**. It provides one place to run detectors, compare them under declared protocols, preserve record/subject-level statistics, and carry hashes, software versions, and provenance alongside results.
|
|
8
|
+
|
|
9
|
+
The repository also ships its own Stage-1 and two-stage Random Forest detector, but that detector is not the project's only or primary claim. The current evidence is deliberately mixed: the two-stage model performs strongly on its locked MIT-BIH split, and two successive model generations have both shown a cross-database drop on INCART. That gap is preserved and documented rather than hidden.
|
|
10
|
+
|
|
11
|
+
## Current evidence
|
|
12
|
+
|
|
13
|
+
**Two distinct model generations exist for the two-stage detector and must not be conflated** (see `validation_reports/VALIDATION_STATUS.md` for the full history). The original locked MIT-BIH row below is the `candidate-features-v3` model from 1.8.1. The current `candidate-features-v4` / windowed-std model now also has explicit MIT-BIH legacy non-regression audits, including the leakage-safe 0.0/0.0 polarity-gate result (F1 0.9644). Those audits do not become clean prospective evidence because MIT-BIH historically informed the adaptive-polarity mechanism.
|
|
14
|
+
|
|
15
|
+
| Protocol | Detector | Model generation | Records | Sensitivity | PPV | F1 |
|
|
16
|
+
|---|---|---|---:|---:|---:|---:|
|
|
17
|
+
| Locked MIT-BIH held-out | ElectroTrace two-stage | v3 (1.8.1 lock) | 12 | 0.9924 | 0.9879 | 0.9902 |
|
|
18
|
+
| MIT-BIH leakage-safe gate audit | ElectroTrace two-stage | v4 (legacy non-regression) | 12 | 0.9390 | 0.9913 | 0.9644 |
|
|
19
|
+
| Locked MIT-BIH held-out | Pan-Tompkins reimplementation | — | 12 | 0.9908 | 0.9954 | 0.9931 |
|
|
20
|
+
| Locked MIT-BIH held-out | Hamilton reimplementation | — | 12 | 0.9990 | 0.9303 | 0.9634 |
|
|
21
|
+
| Locked MIT-BIH held-out | ElectroTrace Stage-1 | — | 12 | 0.9931 | 0.7553 | 0.8580 |
|
|
22
|
+
| INCART external comparison | ElectroTrace two-stage | v3 | 68 | 0.3239 | 0.9679 | 0.4854 |
|
|
23
|
+
| INCART external comparison | ElectroTrace two-stage | v4 (windowed-std) | 68 | 0.9011 | 0.7594 | 0.8242 |
|
|
24
|
+
| INCART certified WFDB comparison | WFDB gqrs | — | 68 | 0.9324 | 0.9265 | 0.9294 |
|
|
25
|
+
| INCART complete source cohort | ElectroTrace two-stage | v4 frozen / development data | 75 | 0.8977 | 0.7633 | 0.8251 |
|
|
26
|
+
| INCART complete source cohort | WFDB gqrs | certified reference baseline | 75 | 0.9372 | 0.9310 | 0.9341 |
|
|
27
|
+
| INCART complete source cohort | WFDB sqrs | certified reference baseline | 75 | 0.7617 | 0.9537 | 0.8469 |
|
|
28
|
+
| European ST-T prospective external | ElectroTrace two-stage | v4 frozen | 90 | 0.9056 | 0.9644 | 0.9341 |
|
|
29
|
+
|
|
30
|
+
These values come from the repository's locked/archived validation artifacts. They are research results, not clinical validation and not evidence of universal detector superiority. **INCART informed v4 feature-scaling development and remains exposed development data**, so neither its historical 68-record row nor the completed 75-record row is a clean generalization estimate. The 75-record completion exists to finish the source cohort reproducibly, including the seven formerly malformed edge annotations under a two-certified-detector repair rule. A later one-shot prospective evaluation on all 90 European ST-T Database records provides independent external evidence for the unchanged frozen v4 configuration (F1 0.9341), but does not make INCART held-out or validate future post-hoc retuning. See `validation_reports/VALIDATION_STATUS.md` and `docs/CROSS_DATABASE_POLICY.md` for the evidence boundaries.
|
|
31
|
+
|
|
32
|
+
## Why ElectroTrace exists
|
|
33
|
+
|
|
34
|
+
Most ECG software answers "what detector should I use?" ElectroTrace is aimed at a different question:
|
|
35
|
+
|
|
36
|
+
> **Under one explicit protocol, how does this detector behave, and can someone else reproduce the answer?**
|
|
37
|
+
|
|
38
|
+
Primary statistics should treat records or subjects as the experimental unit rather than pretending individual beats are independent biological replicates. Validation artifacts can carry dataset hashes, detector configuration, software version, git commit, and protocol metadata.
|
|
39
|
+
|
|
40
|
+
### Primary use cases
|
|
41
|
+
|
|
42
|
+
**Preclinical electrophysiology:** batch animal recordings, retain explicit subject identifiers, and export record/subject-level tables for downstream statistics.
|
|
43
|
+
|
|
44
|
+
**Methods-heavy human ECG research:** evaluate an existing detector on held-out records and retain an auditable artifact for papers, reviews, and lab handoff.
|
|
45
|
+
|
|
46
|
+
**Detector development:** register a detector plugin and compare it against fixed baselines under the same matching rules.
|
|
47
|
+
|
|
48
|
+
ElectroTrace is research software. It is not a clinical device and the current models are not validated for clinical deployment.
|
|
49
|
+
|
|
50
|
+
## Install
|
|
51
|
+
|
|
52
|
+
Python 3.10+ is required.
|
|
53
|
+
|
|
54
|
+
**v1.9.0 is release-ready, but this repository does not claim PyPI availability until the tagged Trusted Publishing workflow succeeds.** Until then, install from source:
|
|
55
|
+
|
|
56
|
+
~~~bash
|
|
57
|
+
git clone https://github.com/Virelion-Biotech/Virelion-ElectroTrace.git
|
|
58
|
+
cd Virelion-ElectroTrace
|
|
59
|
+
python -m pip install -e ".[test,dev]"
|
|
60
|
+
~~~
|
|
61
|
+
|
|
62
|
+
## Five-minute start
|
|
63
|
+
|
|
64
|
+
~~~bash
|
|
65
|
+
electrotrace list
|
|
66
|
+
electrotrace detect recording.edf --detector pan-tompkins --channel 0 -o peaks.csv
|
|
67
|
+
electrotrace batch data/ --detector pan-tompkins --workers 4 -o results/
|
|
68
|
+
electrotrace report validation.json -o validation.html
|
|
69
|
+
~~~
|
|
70
|
+
|
|
71
|
+
Batch outputs:
|
|
72
|
+
|
|
73
|
+
~~~text
|
|
74
|
+
results/
|
|
75
|
+
├── beats.csv
|
|
76
|
+
├── records.csv
|
|
77
|
+
├── subjects.csv
|
|
78
|
+
├── failures.csv
|
|
79
|
+
├── manifest.json
|
|
80
|
+
├── batch_state.json
|
|
81
|
+
└── peaks/
|
|
82
|
+
~~~
|
|
83
|
+
|
|
84
|
+
Subject information is never inferred from filenames. Supply a two-column CSV with record and subject_id using --subject-map when subject aggregation is appropriate.
|
|
85
|
+
|
|
86
|
+
## Validation and benchmarking
|
|
87
|
+
|
|
88
|
+
~~~bash
|
|
89
|
+
electrotrace validate .cache/physionet/mitdb/100 --detector pan-tompkins -o validation.json
|
|
90
|
+
electrotrace bench .cache/physionet/mitdb --detectors pan-tompkins,hamilton --tolerance-ms 75 -o bench.json
|
|
91
|
+
~~~
|
|
92
|
+
|
|
93
|
+
Bench currently operates on local WFDB records. The repository does not vendor PhysioNet data.
|
|
94
|
+
|
|
95
|
+
The frozen experimental QRS boundary locator has also completed a 105-record QT
|
|
96
|
+
Database tolerance-curve study. Reference-centered joint onset/offset success
|
|
97
|
+
was 0.589 at 20 ms, 0.845 at 40 ms, 0.925 at 60 ms, 0.954 at 80 ms, and 0.994
|
|
98
|
+
at 100 ms, with record-level bootstrap uncertainty. This is delineation
|
|
99
|
+
characterization, not an end-to-end detector or clinical-performance claim;
|
|
100
|
+
see `docs/QTDB_VALIDATION.md` and `docs/QRS_DELINEATION.md`.
|
|
101
|
+
|
|
102
|
+
The detector plugin interface is already present. The next benchmark phase will add external package adapters and a regenerated multi-database leaderboard rather than a single scalar ranking.
|
|
103
|
+
|
|
104
|
+
## Model artifacts
|
|
105
|
+
|
|
106
|
+
The two-stage Random Forest is opt-in. New model persistence uses .skops plus a metadata sidecar. Legacy pickle models are migration-only and rejected by default.
|
|
107
|
+
|
|
108
|
+
A model records its feature schema and training scikit-learn major version. Loading stops when the runtime major version is incompatible or when the .skops file contains unknown serialized types.
|
|
109
|
+
|
|
110
|
+
The complete 75-record INCART result is part of the evidence surface: cross-database behavior must be measured instead of inferred from MIT-BIH. The research-use policy for unseen domains is documented in `docs/CROSS_DATABASE_POLICY.md`; ElectroTrace does not silently infer that a new database is in-domain or automatically promote the RF path over established reference baselines.
|
|
111
|
+
|
|
112
|
+
## Validation philosophy
|
|
113
|
+
|
|
114
|
+
- record-level rather than beat-level primary statistics;
|
|
115
|
+
- one-to-one matching under a declared tolerance;
|
|
116
|
+
- locked splits with explicit seeds;
|
|
117
|
+
- bootstrap confidence intervals over records;
|
|
118
|
+
- SHA-256 input hashes and software/git provenance;
|
|
119
|
+
- separate timing error, sensitivity, PPV, and F1;
|
|
120
|
+
- explicit limitations and non-claims.
|
|
121
|
+
|
|
122
|
+
## What is not claimed
|
|
123
|
+
|
|
124
|
+
ElectroTrace does not claim clinical performance, real-time streaming performance, population generalization, or superiority over established QRS detectors.
|
|
125
|
+
|
|
126
|
+
## Documentation and citation
|
|
127
|
+
|
|
128
|
+
See docs/, mkdocs.yml, and CITATION.cff. The repository includes Zenodo metadata for v1.9.0, but a DOI is not claimed until an actual tagged release is archived by Zenodo.
|
|
129
|
+
|
|
130
|
+
## License
|
|
131
|
+
|
|
132
|
+
AGPL-3.0-or-later at present. Any future core/server license split will be an explicit project decision.
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "electrotrace"
|
|
7
|
+
version = "1.9.0"
|
|
8
|
+
description = "Reproducible ECG/electrophysiology annotation, benchmarking, phenotyping, and leakage-safe validation toolkit"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "AGPL-3.0-or-later" }
|
|
11
|
+
requires-python = ">=3.10"
|
|
12
|
+
classifiers = [
|
|
13
|
+
"Development Status :: 4 - Beta",
|
|
14
|
+
"Intended Audience :: Science/Research",
|
|
15
|
+
"License :: OSI Approved :: GNU Affero General Public License v3 or later (AGPLv3+)",
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
18
|
+
"Programming Language :: Python :: 3.10",
|
|
19
|
+
"Programming Language :: Python :: 3.11",
|
|
20
|
+
"Programming Language :: Python :: 3.12",
|
|
21
|
+
"Programming Language :: Python :: 3.13",
|
|
22
|
+
"Topic :: Scientific/Engineering",
|
|
23
|
+
]
|
|
24
|
+
dependencies = [
|
|
25
|
+
"numpy>=1.24",
|
|
26
|
+
"pandas>=2.0",
|
|
27
|
+
"scipy>=1.10",
|
|
28
|
+
"scikit-learn>=1.4",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[project.optional-dependencies]
|
|
32
|
+
wfdb = ["wfdb>=4.1"]
|
|
33
|
+
edf = ["pyedflib>=0.1.38"]
|
|
34
|
+
server = ["flask>=3.0"]
|
|
35
|
+
models = ["skops>=0.14"]
|
|
36
|
+
all = [
|
|
37
|
+
"wfdb>=4.1",
|
|
38
|
+
"pyedflib>=0.1.38",
|
|
39
|
+
"flask>=3.0",
|
|
40
|
+
"skops>=0.14",
|
|
41
|
+
]
|
|
42
|
+
test = [
|
|
43
|
+
"pytest>=8.0",
|
|
44
|
+
"wfdb>=4.1",
|
|
45
|
+
"pyedflib>=0.1.38",
|
|
46
|
+
"skops>=0.14",
|
|
47
|
+
]
|
|
48
|
+
dev = [
|
|
49
|
+
"pytest>=8.0",
|
|
50
|
+
"ruff>=0.6",
|
|
51
|
+
"build>=1.2",
|
|
52
|
+
"mypy>=1.11",
|
|
53
|
+
"wfdb>=4.1",
|
|
54
|
+
"pyedflib>=0.1.38",
|
|
55
|
+
"flask>=3.0",
|
|
56
|
+
"skops>=0.14",
|
|
57
|
+
]
|
|
58
|
+
|
|
59
|
+
[project.scripts]
|
|
60
|
+
electrotrace = "electrotrace.cli:main"
|
|
61
|
+
electrotrace-hearttwin = "electrotrace.hearttwin_adapter:main"
|
|
62
|
+
|
|
63
|
+
[project.entry-points."electrotrace.detectors"]
|
|
64
|
+
# Third-party detector packages register callables here.
|
|
65
|
+
|
|
66
|
+
[project.urls]
|
|
67
|
+
Homepage = "https://github.com/Virelion-Biotech/Virelion-ElectroTrace"
|
|
68
|
+
Documentation = "https://github.com/Virelion-Biotech/Virelion-ElectroTrace/tree/main/docs"
|
|
69
|
+
Repository = "https://github.com/Virelion-Biotech/Virelion-ElectroTrace"
|
|
70
|
+
Issues = "https://github.com/Virelion-Biotech/Virelion-ElectroTrace/issues"
|
|
71
|
+
|
|
72
|
+
[tool.setuptools.packages.find]
|
|
73
|
+
where = ["src"]
|
|
74
|
+
|
|
75
|
+
[tool.pytest.ini_options]
|
|
76
|
+
pythonpath = ["src", "."]
|
|
77
|
+
testpaths = ["tests"]
|
|
78
|
+
addopts = "--strict-markers"
|
|
79
|
+
|
|
80
|
+
[tool.ruff]
|
|
81
|
+
line-length = 120
|
|
82
|
+
target-version = "py310"
|
|
83
|
+
|
|
84
|
+
[tool.ruff.lint]
|
|
85
|
+
select = ["E", "F", "I", "UP"]
|
|
86
|
+
|
|
87
|
+
[tool.ruff.lint.isort]
|
|
88
|
+
known-first-party = ["electrotrace", "scripts"]
|
|
89
|
+
|
|
90
|
+
[tool.mypy]
|
|
91
|
+
python_version = "3.10"
|
|
92
|
+
ignore_missing_imports = true
|
|
93
|
+
check_untyped_defs = true
|
|
94
|
+
warn_unused_ignores = true
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""ElectroTrace: reproducible ECG/electrophysiology research and benchmarking tools."""
|
|
2
|
+
|
|
3
|
+
__version__ = "1.9.0"
|
|
4
|
+
|
|
5
|
+
from .candidate_suppressor import CandidateSuppressor
|
|
6
|
+
from .io import load_recording
|
|
7
|
+
from .provenance import DatasetManifest, manifest_from_dict
|
|
8
|
+
from .qrs_delineation import delineate_qrs
|
|
9
|
+
from .research_validation import build_validation_report, summarize_records_rigorous, write_validation_report
|
|
10
|
+
from .signal import apply_pipeline
|
|
11
|
+
from .validation import validate_record
|
|
12
|
+
from .validation_detectors import detect_r_peaks, detect_r_peaks_two_stage
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"__version__",
|
|
16
|
+
"CandidateSuppressor",
|
|
17
|
+
"DatasetManifest",
|
|
18
|
+
"manifest_from_dict",
|
|
19
|
+
"load_recording",
|
|
20
|
+
"apply_pipeline",
|
|
21
|
+
"detect_r_peaks",
|
|
22
|
+
"detect_r_peaks_two_stage",
|
|
23
|
+
"delineate_qrs",
|
|
24
|
+
"validate_record",
|
|
25
|
+
"build_validation_report",
|
|
26
|
+
"summarize_records_rigorous",
|
|
27
|
+
"write_validation_report",
|
|
28
|
+
]
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"""Annotation model, validation, serialization, review state, and agreement metrics."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
import math
|
|
6
|
+
import uuid
|
|
7
|
+
from dataclasses import asdict, dataclass, field
|
|
8
|
+
from typing import Literal, Optional
|
|
9
|
+
|
|
10
|
+
AnnotationType = Literal["interval", "point"]
|
|
11
|
+
ReviewStatus = Literal["unreviewed", "accepted", "flagged"]
|
|
12
|
+
DEFAULT_LABELS = ["P Wave", "QRS", "T Wave", "Pacemaker Spike", "R Peak", "Artifact", "Abnormal Beat", "Graft Activation", "Arrhythmia", "Custom"]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class Annotation:
|
|
17
|
+
label: str
|
|
18
|
+
type: AnnotationType
|
|
19
|
+
channel: str
|
|
20
|
+
start: Optional[float] = None
|
|
21
|
+
end: Optional[float] = None
|
|
22
|
+
time: Optional[float] = None
|
|
23
|
+
confidence: float = 1.0
|
|
24
|
+
notes: str = ""
|
|
25
|
+
annotator: str = ""
|
|
26
|
+
status: ReviewStatus = "unreviewed"
|
|
27
|
+
reviewer: str = ""
|
|
28
|
+
review_notes: str = ""
|
|
29
|
+
id: str = field(default_factory=lambda: uuid.uuid4().hex[:10])
|
|
30
|
+
|
|
31
|
+
def validate(self, duration_s: float | None = None, start_time_s: float = 0.0, end_time_s: float | None = None) -> None:
|
|
32
|
+
if not self.label.strip():
|
|
33
|
+
raise ValueError("label must not be empty")
|
|
34
|
+
if not self.channel.strip():
|
|
35
|
+
raise ValueError("channel must not be empty")
|
|
36
|
+
if not math.isfinite(float(self.confidence)) or not 0.0 <= self.confidence <= 1.0:
|
|
37
|
+
raise ValueError("confidence must be between 0 and 1")
|
|
38
|
+
if self.status not in {"unreviewed", "accepted", "flagged"}:
|
|
39
|
+
raise ValueError("invalid review status")
|
|
40
|
+
start_bound = float(start_time_s)
|
|
41
|
+
end_bound = float(end_time_s) if end_time_s is not None else (start_bound + float(duration_s) if duration_s is not None else None)
|
|
42
|
+
if not math.isfinite(start_bound) or (end_bound is not None and not math.isfinite(end_bound)) or (end_bound is not None and end_bound < start_bound):
|
|
43
|
+
raise ValueError("invalid recording time bounds")
|
|
44
|
+
if self.type == "interval":
|
|
45
|
+
if self.start is None or self.end is None:
|
|
46
|
+
raise ValueError("interval annotations require start and end")
|
|
47
|
+
if not math.isfinite(self.start) or not math.isfinite(self.end):
|
|
48
|
+
raise ValueError("interval coordinates must be finite")
|
|
49
|
+
if self.end <= self.start:
|
|
50
|
+
raise ValueError("end must be greater than start")
|
|
51
|
+
if self.time is not None:
|
|
52
|
+
raise ValueError("interval annotations must not define time")
|
|
53
|
+
if end_bound is not None and (self.start < start_bound or self.end > end_bound):
|
|
54
|
+
raise ValueError("interval is outside the recording bounds")
|
|
55
|
+
elif self.type == "point":
|
|
56
|
+
if self.time is None:
|
|
57
|
+
raise ValueError("point annotations require time")
|
|
58
|
+
if not math.isfinite(self.time):
|
|
59
|
+
raise ValueError("point time must be finite")
|
|
60
|
+
if self.start is not None or self.end is not None:
|
|
61
|
+
raise ValueError("point annotations must not define start/end")
|
|
62
|
+
if end_bound is not None and not start_bound <= self.time <= end_bound:
|
|
63
|
+
raise ValueError("point is outside the recording bounds")
|
|
64
|
+
else:
|
|
65
|
+
raise ValueError(f"unknown annotation type: {self.type}")
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def position(self) -> float:
|
|
69
|
+
return self.start if self.type == "interval" else self.time # type: ignore[return-value]
|
|
70
|
+
|
|
71
|
+
def to_dict(self) -> dict:
|
|
72
|
+
return asdict(self)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class AnnotationStore:
|
|
76
|
+
def __init__(self, duration_s: float | None = None, start_time_s: float = 0.0, end_time_s: float | None = None):
|
|
77
|
+
self.duration_s = duration_s
|
|
78
|
+
self.start_time_s = float(start_time_s)
|
|
79
|
+
self.end_time_s = float(end_time_s) if end_time_s is not None else (self.start_time_s + float(duration_s) if duration_s is not None else None)
|
|
80
|
+
self._items: list[Annotation] = []
|
|
81
|
+
|
|
82
|
+
@property
|
|
83
|
+
def items(self) -> list[Annotation]:
|
|
84
|
+
return list(self._items)
|
|
85
|
+
|
|
86
|
+
def add(self, ann: Annotation) -> Annotation:
|
|
87
|
+
ann.validate(self.duration_s, self.start_time_s, self.end_time_s)
|
|
88
|
+
if any(a.id == ann.id for a in self._items):
|
|
89
|
+
raise ValueError(f"duplicate annotation id: {ann.id}")
|
|
90
|
+
self._items.append(ann)
|
|
91
|
+
self._sort()
|
|
92
|
+
return ann
|
|
93
|
+
|
|
94
|
+
def update(self, ann_id: str, **changes) -> Annotation:
|
|
95
|
+
for idx, existing in enumerate(self._items):
|
|
96
|
+
if existing.id == ann_id:
|
|
97
|
+
data = {**asdict(existing), **changes}
|
|
98
|
+
updated = Annotation(**data)
|
|
99
|
+
updated.validate(self.duration_s, self.start_time_s, self.end_time_s)
|
|
100
|
+
self._items[idx] = updated
|
|
101
|
+
self._sort()
|
|
102
|
+
return updated
|
|
103
|
+
raise KeyError(ann_id)
|
|
104
|
+
|
|
105
|
+
def delete(self, ann_id: str) -> bool:
|
|
106
|
+
before = len(self._items)
|
|
107
|
+
self._items = [a for a in self._items if a.id != ann_id]
|
|
108
|
+
return len(self._items) != before
|
|
109
|
+
|
|
110
|
+
def duplicate(self, ann_id: str) -> Annotation:
|
|
111
|
+
for existing in self._items:
|
|
112
|
+
if existing.id == ann_id:
|
|
113
|
+
data = asdict(existing)
|
|
114
|
+
data["id"] = uuid.uuid4().hex[:10]
|
|
115
|
+
data["status"] = "unreviewed"
|
|
116
|
+
data["reviewer"] = ""
|
|
117
|
+
data["review_notes"] = ""
|
|
118
|
+
return self.add(Annotation(**data))
|
|
119
|
+
raise KeyError(ann_id)
|
|
120
|
+
|
|
121
|
+
def clear(self) -> None:
|
|
122
|
+
self._items.clear()
|
|
123
|
+
|
|
124
|
+
def _sort(self) -> None:
|
|
125
|
+
self._items.sort(key=lambda a: a.position)
|
|
126
|
+
|
|
127
|
+
def to_dict(self, source_file: str = "", metadata: dict | None = None) -> dict:
|
|
128
|
+
meta = dict(metadata or {})
|
|
129
|
+
meta.setdefault("time_start_s", self.start_time_s)
|
|
130
|
+
if self.end_time_s is not None:
|
|
131
|
+
meta.setdefault("time_end_s", self.end_time_s)
|
|
132
|
+
return {"schema": "electrotrace.annotation/v2", "file": source_file, "metadata": meta, "annotations": [a.to_dict() for a in self._items]}
|
|
133
|
+
|
|
134
|
+
def to_json(self, source_file: str = "", metadata: dict | None = None) -> str:
|
|
135
|
+
return json.dumps(self.to_dict(source_file, metadata), indent=2)
|
|
136
|
+
|
|
137
|
+
@classmethod
|
|
138
|
+
def from_dict(cls, data: dict, duration_s: float | None = None, start_time_s: float | None = None, end_time_s: float | None = None) -> "AnnotationStore":
|
|
139
|
+
metadata = data.get("metadata") or {}
|
|
140
|
+
start = float(metadata.get("time_start_s", 0.0) if start_time_s is None else start_time_s)
|
|
141
|
+
end = metadata.get("time_end_s") if end_time_s is None else end_time_s
|
|
142
|
+
store = cls(duration_s=duration_s, start_time_s=start, end_time_s=float(end) if end is not None else None)
|
|
143
|
+
schema = data.get("schema", data.get("annotation_schema", ""))
|
|
144
|
+
if schema and schema not in {"electrotrace.annotation/v2", "v1"}:
|
|
145
|
+
raise ValueError(f"unsupported annotation schema: {schema}")
|
|
146
|
+
for raw in data.get("annotations", []):
|
|
147
|
+
store.add(Annotation(**raw))
|
|
148
|
+
return store
|
|
149
|
+
|
|
150
|
+
@classmethod
|
|
151
|
+
def from_json(cls, text: str, duration_s: float | None = None, start_time_s: float | None = None, end_time_s: float | None = None) -> "AnnotationStore":
|
|
152
|
+
try:
|
|
153
|
+
data = json.loads(text)
|
|
154
|
+
except json.JSONDecodeError as exc:
|
|
155
|
+
raise ValueError(f"invalid JSON: {exc}") from exc
|
|
156
|
+
return cls.from_dict(data, duration_s=duration_s, start_time_s=start_time_s, end_time_s=end_time_s)
|
|
157
|
+
|
|
158
|
+
def to_csv_rows(self, source_file: str = "") -> list[dict]:
|
|
159
|
+
return [{"file": source_file, "id": a.id, "type": a.type, "label": a.label, "channel": a.channel, "start": a.start, "end": a.end, "time": a.time, "confidence": a.confidence, "notes": a.notes, "annotator": a.annotator, "status": a.status, "reviewer": a.reviewer, "review_notes": a.review_notes} for a in self._items]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def point_agreement(a: list[Annotation], b: list[Annotation], tolerance_s: float = 0.04) -> dict:
|
|
163
|
+
if tolerance_s <= 0 or not math.isfinite(tolerance_s):
|
|
164
|
+
raise ValueError("tolerance_s must be positive and finite")
|
|
165
|
+
left = [x for x in a if x.type == "point"]
|
|
166
|
+
right = [x for x in b if x.type == "point"]
|
|
167
|
+
used: set[str] = set()
|
|
168
|
+
errors = []
|
|
169
|
+
matches = 0
|
|
170
|
+
for x in left:
|
|
171
|
+
candidates = [
|
|
172
|
+
y for y in right
|
|
173
|
+
if y.id not in used
|
|
174
|
+
and y.label == x.label
|
|
175
|
+
and y.channel == x.channel
|
|
176
|
+
and y.time is not None
|
|
177
|
+
and x.time is not None
|
|
178
|
+
and abs(y.time - x.time) <= tolerance_s
|
|
179
|
+
]
|
|
180
|
+
if candidates:
|
|
181
|
+
y = min(candidates, key=lambda z: abs(z.time - x.time))
|
|
182
|
+
used.add(y.id)
|
|
183
|
+
matches += 1
|
|
184
|
+
errors.append(abs(y.time - x.time))
|
|
185
|
+
total = max(len(left), len(right), 1)
|
|
186
|
+
return {"matches": matches, "agreement_rate": matches / total, "mean_absolute_error_s": sum(errors) / len(errors) if errors else None}
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def interval_iou(a: Annotation, b: Annotation) -> float:
|
|
190
|
+
if a.type != "interval" or b.type != "interval":
|
|
191
|
+
return 0.0
|
|
192
|
+
inter = max(0.0, min(a.end, b.end) - max(a.start, b.start)) # type: ignore[arg-type]
|
|
193
|
+
union = max(a.end, b.end) - min(a.start, b.start) # type: ignore[arg-type]
|
|
194
|
+
return inter / union if union else 0.0
|