electrotrace 1.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. electrotrace-1.9.0/LICENSE +20 -0
  2. electrotrace-1.9.0/PKG-INFO +187 -0
  3. electrotrace-1.9.0/README.md +132 -0
  4. electrotrace-1.9.0/pyproject.toml +94 -0
  5. electrotrace-1.9.0/setup.cfg +4 -0
  6. electrotrace-1.9.0/src/electrotrace/__init__.py +28 -0
  7. electrotrace-1.9.0/src/electrotrace/__main__.py +3 -0
  8. electrotrace-1.9.0/src/electrotrace/annotations.py +194 -0
  9. electrotrace-1.9.0/src/electrotrace/baseline_detectors.py +122 -0
  10. electrotrace-1.9.0/src/electrotrace/beats.py +47 -0
  11. electrotrace-1.9.0/src/electrotrace/benchmark.py +154 -0
  12. electrotrace-1.9.0/src/electrotrace/candidate_suppressor.py +441 -0
  13. electrotrace-1.9.0/src/electrotrace/cli.py +630 -0
  14. electrotrace-1.9.0/src/electrotrace/detectors.py +152 -0
  15. electrotrace-1.9.0/src/electrotrace/formats.py +147 -0
  16. electrotrace-1.9.0/src/electrotrace/fp_analysis.py +358 -0
  17. electrotrace-1.9.0/src/electrotrace/hearttwin_adapter.py +90 -0
  18. electrotrace-1.9.0/src/electrotrace/io.py +163 -0
  19. electrotrace-1.9.0/src/electrotrace/lead_quality.py +138 -0
  20. electrotrace-1.9.0/src/electrotrace/lead_selection.py +169 -0
  21. electrotrace-1.9.0/src/electrotrace/metadata.py +111 -0
  22. electrotrace-1.9.0/src/electrotrace/ml.py +178 -0
  23. electrotrace-1.9.0/src/electrotrace/phenotype.py +84 -0
  24. electrotrace-1.9.0/src/electrotrace/phenotype_validation.py +133 -0
  25. electrotrace-1.9.0/src/electrotrace/polarity_v2.py +201 -0
  26. electrotrace-1.9.0/src/electrotrace/project.py +43 -0
  27. electrotrace-1.9.0/src/electrotrace/project_store.py +147 -0
  28. electrotrace-1.9.0/src/electrotrace/provenance.py +155 -0
  29. electrotrace-1.9.0/src/electrotrace/qrs_delineation.py +139 -0
  30. electrotrace-1.9.0/src/electrotrace/qtdb_detector_adapter.py +7 -0
  31. electrotrace-1.9.0/src/electrotrace/research_validation.py +146 -0
  32. electrotrace-1.9.0/src/electrotrace/scale_estimation.py +172 -0
  33. electrotrace-1.9.0/src/electrotrace/security.py +57 -0
  34. electrotrace-1.9.0/src/electrotrace/server_app.py +375 -0
  35. electrotrace-1.9.0/src/electrotrace/signal.py +84 -0
  36. electrotrace-1.9.0/src/electrotrace/statistics.py +98 -0
  37. electrotrace-1.9.0/src/electrotrace/threshold_selection.py +40 -0
  38. electrotrace-1.9.0/src/electrotrace/validation.py +241 -0
  39. electrotrace-1.9.0/src/electrotrace/validation_detectors.py +570 -0
  40. electrotrace-1.9.0/src/electrotrace/wfdb_records.py +375 -0
  41. electrotrace-1.9.0/src/electrotrace/window.py +117 -0
  42. electrotrace-1.9.0/src/electrotrace.egg-info/PKG-INFO +187 -0
  43. electrotrace-1.9.0/src/electrotrace.egg-info/SOURCES.txt +106 -0
  44. electrotrace-1.9.0/src/electrotrace.egg-info/dependency_links.txt +1 -0
  45. electrotrace-1.9.0/src/electrotrace.egg-info/entry_points.txt +3 -0
  46. electrotrace-1.9.0/src/electrotrace.egg-info/requires.txt +38 -0
  47. electrotrace-1.9.0/src/electrotrace.egg-info/top_level.txt +1 -0
  48. electrotrace-1.9.0/tests/test_annotations.py +54 -0
  49. electrotrace-1.9.0/tests/test_api.py +96 -0
  50. electrotrace-1.9.0/tests/test_audit_calibration.py +18 -0
  51. electrotrace-1.9.0/tests/test_audit_hardening.py +38 -0
  52. electrotrace-1.9.0/tests/test_audit_regressions.py +65 -0
  53. electrotrace-1.9.0/tests/test_baseline_detectors.py +21 -0
  54. electrotrace-1.9.0/tests/test_beats_ml.py +64 -0
  55. electrotrace-1.9.0/tests/test_candidate_suppressor.py +75 -0
  56. electrotrace-1.9.0/tests/test_certified_wfdb_mitdb_locked.py +131 -0
  57. electrotrace-1.9.0/tests/test_derive_polarity_thresholds_mitdb_extended.py +308 -0
  58. electrotrace-1.9.0/tests/test_detector_registry.py +12 -0
  59. electrotrace-1.9.0/tests/test_diagnose_merge_incart.py +34 -0
  60. electrotrace-1.9.0/tests/test_dual_polarity_merge.py +267 -0
  61. electrotrace-1.9.0/tests/test_edb_lead_selector_development.py +72 -0
  62. electrotrace-1.9.0/tests/test_edb_posthoc_failures.py +140 -0
  63. electrotrace-1.9.0/tests/test_edb_prospective.py +241 -0
  64. electrotrace-1.9.0/tests/test_evaluate_v2_gate.py +427 -0
  65. electrotrace-1.9.0/tests/test_fp_analysis.py +149 -0
  66. electrotrace-1.9.0/tests/test_hardening.py +44 -0
  67. electrotrace-1.9.0/tests/test_hearttwin_adapter.py +23 -0
  68. electrotrace-1.9.0/tests/test_incart_complete_characterization.py +144 -0
  69. electrotrace-1.9.0/tests/test_incart_forensics_scripts.py +187 -0
  70. electrotrace-1.9.0/tests/test_io.py +31 -0
  71. electrotrace-1.9.0/tests/test_lead_quality.py +139 -0
  72. electrotrace-1.9.0/tests/test_lead_selection.py +90 -0
  73. electrotrace-1.9.0/tests/test_lead_selection_v3.py +40 -0
  74. electrotrace-1.9.0/tests/test_ltafdb_label_free_features.py +87 -0
  75. electrotrace-1.9.0/tests/test_ltafdb_lead_selector_prospective.py +224 -0
  76. electrotrace-1.9.0/tests/test_ltafdb_posthoc_selector_audit.py +62 -0
  77. electrotrace-1.9.0/tests/test_ltstdb_zymed_selector_v3_posthoc.py +137 -0
  78. electrotrace-1.9.0/tests/test_ltstdb_zymed_selector_v3_prospective.py +124 -0
  79. electrotrace-1.9.0/tests/test_metadata.py +32 -0
  80. electrotrace-1.9.0/tests/test_model_persistence.py +55 -0
  81. electrotrace-1.9.0/tests/test_phenotype_validation.py +32 -0
  82. electrotrace-1.9.0/tests/test_polarity_v2.py +72 -0
  83. electrotrace-1.9.0/tests/test_polarity_width_override.py +104 -0
  84. electrotrace-1.9.0/tests/test_probability_alignment_bugfix.py +95 -0
  85. electrotrace-1.9.0/tests/test_project_api.py +21 -0
  86. electrotrace-1.9.0/tests/test_project_store_window.py +28 -0
  87. electrotrace-1.9.0/tests/test_provenance_and_research_validation.py +135 -0
  88. electrotrace-1.9.0/tests/test_public_api.py +28 -0
  89. electrotrace-1.9.0/tests/test_qrs_delineation.py +39 -0
  90. electrotrace-1.9.0/tests/test_qtdb_tolerance_curve.py +135 -0
  91. electrotrace-1.9.0/tests/test_recalibration_and_inspection.py +538 -0
  92. electrotrace-1.9.0/tests/test_record_calibration.py +26 -0
  93. electrotrace-1.9.0/tests/test_recovery.py +36 -0
  94. electrotrace-1.9.0/tests/test_release_metadata.py +47 -0
  95. electrotrace-1.9.0/tests/test_research.py +61 -0
  96. electrotrace-1.9.0/tests/test_security_and_project_store.py +40 -0
  97. electrotrace-1.9.0/tests/test_signal.py +27 -0
  98. electrotrace-1.9.0/tests/test_stage1_scale_estimator.py +174 -0
  99. electrotrace-1.9.0/tests/test_stage2_feature_scale.py +77 -0
  100. electrotrace-1.9.0/tests/test_svdb_selector_v2_posthoc.py +75 -0
  101. electrotrace-1.9.0/tests/test_svdb_selector_v2_prospective.py +298 -0
  102. electrotrace-1.9.0/tests/test_two_stage_detector.py +63 -0
  103. electrotrace-1.9.0/tests/test_v2_gate_confidence.py +152 -0
  104. electrotrace-1.9.0/tests/test_validation.py +65 -0
  105. electrotrace-1.9.0/tests/test_validation_detector.py +19 -0
  106. electrotrace-1.9.0/tests/test_validation_detectors.py +35 -0
  107. electrotrace-1.9.0/tests/test_wfdb_records.py +238 -0
  108. electrotrace-1.9.0/tests/test_window.py +33 -0
@@ -0,0 +1,20 @@
1
+ GNU AFFERO GENERAL PUBLIC LICENSE
2
+ Version 3, 19 November 2007
3
+
4
+ Copyright (c) 2026 Virelion Biotech
5
+
6
+ This program is free software: you can redistribute it and/or modify
7
+ it under the terms of the GNU Affero General Public License as published by
8
+ the Free Software Foundation, either version 3 of the License, or
9
+ (at your option) any later version.
10
+
11
+ This program is distributed in the hope that it will be useful,
12
+ but WITHOUT ANY WARRANTY; without even the implied warranty of
13
+ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
14
+ GNU Affero General Public License for more details.
15
+
16
+ You should have received a copy of the GNU Affero General Public License
17
+ along with this program. If not, see <https://www.gnu.org/licenses/>.
18
+
19
+ For the full license text, see:
20
+ https://www.gnu.org/licenses/agpl-3.0.html
@@ -0,0 +1,187 @@
1
+ Metadata-Version: 2.4
2
+ Name: electrotrace
3
+ Version: 1.9.0
4
+ Summary: Reproducible ECG/electrophysiology annotation, benchmarking, phenotyping, and leakage-safe validation toolkit
5
+ License: AGPL-3.0-or-later
6
+ Project-URL: Homepage, https://github.com/Virelion-Biotech/Virelion-ElectroTrace
7
+ Project-URL: Documentation, https://github.com/Virelion-Biotech/Virelion-ElectroTrace/tree/main/docs
8
+ Project-URL: Repository, https://github.com/Virelion-Biotech/Virelion-ElectroTrace
9
+ Project-URL: Issues, https://github.com/Virelion-Biotech/Virelion-ElectroTrace/issues
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: License :: OSI Approved :: GNU Affero General Public License v3 or later (AGPLv3+)
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3 :: Only
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Scientific/Engineering
20
+ Requires-Python: >=3.10
21
+ Description-Content-Type: text/markdown
22
+ License-File: LICENSE
23
+ Requires-Dist: numpy>=1.24
24
+ Requires-Dist: pandas>=2.0
25
+ Requires-Dist: scipy>=1.10
26
+ Requires-Dist: scikit-learn>=1.4
27
+ Provides-Extra: wfdb
28
+ Requires-Dist: wfdb>=4.1; extra == "wfdb"
29
+ Provides-Extra: edf
30
+ Requires-Dist: pyedflib>=0.1.38; extra == "edf"
31
+ Provides-Extra: server
32
+ Requires-Dist: flask>=3.0; extra == "server"
33
+ Provides-Extra: models
34
+ Requires-Dist: skops>=0.14; extra == "models"
35
+ Provides-Extra: all
36
+ Requires-Dist: wfdb>=4.1; extra == "all"
37
+ Requires-Dist: pyedflib>=0.1.38; extra == "all"
38
+ Requires-Dist: flask>=3.0; extra == "all"
39
+ Requires-Dist: skops>=0.14; extra == "all"
40
+ Provides-Extra: test
41
+ Requires-Dist: pytest>=8.0; extra == "test"
42
+ Requires-Dist: wfdb>=4.1; extra == "test"
43
+ Requires-Dist: pyedflib>=0.1.38; extra == "test"
44
+ Requires-Dist: skops>=0.14; extra == "test"
45
+ Provides-Extra: dev
46
+ Requires-Dist: pytest>=8.0; extra == "dev"
47
+ Requires-Dist: ruff>=0.6; extra == "dev"
48
+ Requires-Dist: build>=1.2; extra == "dev"
49
+ Requires-Dist: mypy>=1.11; extra == "dev"
50
+ Requires-Dist: wfdb>=4.1; extra == "dev"
51
+ Requires-Dist: pyedflib>=0.1.38; extra == "dev"
52
+ Requires-Dist: flask>=3.0; extra == "dev"
53
+ Requires-Dist: skops>=0.14; extra == "dev"
54
+ Dynamic: license-file
55
+
56
+ # ElectroTrace
57
+
58
+ **Current source version:** 1.9.0 (release candidate)
59
+
60
+ **A reproducible ECG/electrophysiology annotation and benchmarking workbench.**
61
+
62
+ ElectroTrace is built around a simple principle: **the evidence is the product**. It provides one place to run detectors, compare them under declared protocols, preserve record/subject-level statistics, and carry hashes, software versions, and provenance alongside results.
63
+
64
+ The repository also ships its own Stage-1 and two-stage Random Forest detector, but that detector is not the project's only or primary claim. The current evidence is deliberately mixed: the two-stage model performs strongly on its locked MIT-BIH split, and two successive model generations have both shown a cross-database drop on INCART. That gap is preserved and documented rather than hidden.
65
+
66
+ ## Current evidence
67
+
68
+ **Two distinct model generations exist for the two-stage detector and must not be conflated** (see `validation_reports/VALIDATION_STATUS.md` for the full history). The original locked MIT-BIH row below is the `candidate-features-v3` model from 1.8.1. The current `candidate-features-v4` / windowed-std model now also has explicit MIT-BIH legacy non-regression audits, including the leakage-safe 0.0/0.0 polarity-gate result (F1 0.9644). Those audits do not become clean prospective evidence because MIT-BIH historically informed the adaptive-polarity mechanism.
69
+
70
+ | Protocol | Detector | Model generation | Records | Sensitivity | PPV | F1 |
71
+ |---|---|---|---:|---:|---:|---:|
72
+ | Locked MIT-BIH held-out | ElectroTrace two-stage | v3 (1.8.1 lock) | 12 | 0.9924 | 0.9879 | 0.9902 |
73
+ | MIT-BIH leakage-safe gate audit | ElectroTrace two-stage | v4 (legacy non-regression) | 12 | 0.9390 | 0.9913 | 0.9644 |
74
+ | Locked MIT-BIH held-out | Pan-Tompkins reimplementation | — | 12 | 0.9908 | 0.9954 | 0.9931 |
75
+ | Locked MIT-BIH held-out | Hamilton reimplementation | — | 12 | 0.9990 | 0.9303 | 0.9634 |
76
+ | Locked MIT-BIH held-out | ElectroTrace Stage-1 | — | 12 | 0.9931 | 0.7553 | 0.8580 |
77
+ | INCART external comparison | ElectroTrace two-stage | v3 | 68 | 0.3239 | 0.9679 | 0.4854 |
78
+ | INCART external comparison | ElectroTrace two-stage | v4 (windowed-std) | 68 | 0.9011 | 0.7594 | 0.8242 |
79
+ | INCART certified WFDB comparison | WFDB gqrs | — | 68 | 0.9324 | 0.9265 | 0.9294 |
80
+ | INCART complete source cohort | ElectroTrace two-stage | v4 frozen / development data | 75 | 0.8977 | 0.7633 | 0.8251 |
81
+ | INCART complete source cohort | WFDB gqrs | certified reference baseline | 75 | 0.9372 | 0.9310 | 0.9341 |
82
+ | INCART complete source cohort | WFDB sqrs | certified reference baseline | 75 | 0.7617 | 0.9537 | 0.8469 |
83
+ | European ST-T prospective external | ElectroTrace two-stage | v4 frozen | 90 | 0.9056 | 0.9644 | 0.9341 |
84
+
85
+ These values come from the repository's locked/archived validation artifacts. They are research results, not clinical validation and not evidence of universal detector superiority. **INCART informed v4 feature-scaling development and remains exposed development data**, so neither its historical 68-record row nor the completed 75-record row is a clean generalization estimate. The 75-record completion exists to finish the source cohort reproducibly, including the seven formerly malformed edge annotations under a two-certified-detector repair rule. A later one-shot prospective evaluation on all 90 European ST-T Database records provides independent external evidence for the unchanged frozen v4 configuration (F1 0.9341), but does not make INCART held-out or validate future post-hoc retuning. See `validation_reports/VALIDATION_STATUS.md` and `docs/CROSS_DATABASE_POLICY.md` for the evidence boundaries.
86
+
87
+ ## Why ElectroTrace exists
88
+
89
+ Most ECG software answers "what detector should I use?" ElectroTrace is aimed at a different question:
90
+
91
+ > **Under one explicit protocol, how does this detector behave, and can someone else reproduce the answer?**
92
+
93
+ Primary statistics should treat records or subjects as the experimental unit rather than pretending individual beats are independent biological replicates. Validation artifacts can carry dataset hashes, detector configuration, software version, git commit, and protocol metadata.
94
+
95
+ ### Primary use cases
96
+
97
+ **Preclinical electrophysiology:** batch animal recordings, retain explicit subject identifiers, and export record/subject-level tables for downstream statistics.
98
+
99
+ **Methods-heavy human ECG research:** evaluate an existing detector on held-out records and retain an auditable artifact for papers, reviews, and lab handoff.
100
+
101
+ **Detector development:** register a detector plugin and compare it against fixed baselines under the same matching rules.
102
+
103
+ ElectroTrace is research software. It is not a clinical device and the current models are not validated for clinical deployment.
104
+
105
+ ## Install
106
+
107
+ Python 3.10+ is required.
108
+
109
+ **v1.9.0 is release-ready, but this repository does not claim PyPI availability until the tagged Trusted Publishing workflow succeeds.** Until then, install from source:
110
+
111
+ ~~~bash
112
+ git clone https://github.com/Virelion-Biotech/Virelion-ElectroTrace.git
113
+ cd Virelion-ElectroTrace
114
+ python -m pip install -e ".[test,dev]"
115
+ ~~~
116
+
117
+ ## Five-minute start
118
+
119
+ ~~~bash
120
+ electrotrace list
121
+ electrotrace detect recording.edf --detector pan-tompkins --channel 0 -o peaks.csv
122
+ electrotrace batch data/ --detector pan-tompkins --workers 4 -o results/
123
+ electrotrace report validation.json -o validation.html
124
+ ~~~
125
+
126
+ Batch outputs:
127
+
128
+ ~~~text
129
+ results/
130
+ ├── beats.csv
131
+ ├── records.csv
132
+ ├── subjects.csv
133
+ ├── failures.csv
134
+ ├── manifest.json
135
+ ├── batch_state.json
136
+ └── peaks/
137
+ ~~~
138
+
139
+ Subject information is never inferred from filenames. Supply a two-column CSV with record and subject_id using --subject-map when subject aggregation is appropriate.
140
+
141
+ ## Validation and benchmarking
142
+
143
+ ~~~bash
144
+ electrotrace validate .cache/physionet/mitdb/100 --detector pan-tompkins -o validation.json
145
+ electrotrace bench .cache/physionet/mitdb --detectors pan-tompkins,hamilton --tolerance-ms 75 -o bench.json
146
+ ~~~
147
+
148
+ Bench currently operates on local WFDB records. The repository does not vendor PhysioNet data.
149
+
150
+ The frozen experimental QRS boundary locator has also completed a 105-record QT
151
+ Database tolerance-curve study. Reference-centered joint onset/offset success
152
+ was 0.589 at 20 ms, 0.845 at 40 ms, 0.925 at 60 ms, 0.954 at 80 ms, and 0.994
153
+ at 100 ms, with record-level bootstrap uncertainty. This is delineation
154
+ characterization, not an end-to-end detector or clinical-performance claim;
155
+ see `docs/QTDB_VALIDATION.md` and `docs/QRS_DELINEATION.md`.
156
+
157
+ The detector plugin interface is already present. The next benchmark phase will add external package adapters and a regenerated multi-database leaderboard rather than a single scalar ranking.
158
+
159
+ ## Model artifacts
160
+
161
+ The two-stage Random Forest is opt-in. New model persistence uses .skops plus a metadata sidecar. Legacy pickle models are migration-only and rejected by default.
162
+
163
+ A model records its feature schema and training scikit-learn major version. Loading stops when the runtime major version is incompatible or when the .skops file contains unknown serialized types.
164
+
165
+ The complete 75-record INCART result is part of the evidence surface: cross-database behavior must be measured instead of inferred from MIT-BIH. The research-use policy for unseen domains is documented in `docs/CROSS_DATABASE_POLICY.md`; ElectroTrace does not silently infer that a new database is in-domain or automatically promote the RF path over established reference baselines.
166
+
167
+ ## Validation philosophy
168
+
169
+ - record-level rather than beat-level primary statistics;
170
+ - one-to-one matching under a declared tolerance;
171
+ - locked splits with explicit seeds;
172
+ - bootstrap confidence intervals over records;
173
+ - SHA-256 input hashes and software/git provenance;
174
+ - separate timing error, sensitivity, PPV, and F1;
175
+ - explicit limitations and non-claims.
176
+
177
+ ## What is not claimed
178
+
179
+ ElectroTrace does not claim clinical performance, real-time streaming performance, population generalization, or superiority over established QRS detectors.
180
+
181
+ ## Documentation and citation
182
+
183
+ See docs/, mkdocs.yml, and CITATION.cff. The repository includes Zenodo metadata for v1.9.0, but a DOI is not claimed until an actual tagged release is archived by Zenodo.
184
+
185
+ ## License
186
+
187
+ AGPL-3.0-or-later at present. Any future core/server license split will be an explicit project decision.
@@ -0,0 +1,132 @@
1
+ # ElectroTrace
2
+
3
+ **Current source version:** 1.9.0 (release candidate)
4
+
5
+ **A reproducible ECG/electrophysiology annotation and benchmarking workbench.**
6
+
7
+ ElectroTrace is built around a simple principle: **the evidence is the product**. It provides one place to run detectors, compare them under declared protocols, preserve record/subject-level statistics, and carry hashes, software versions, and provenance alongside results.
8
+
9
+ The repository also ships its own Stage-1 and two-stage Random Forest detector, but that detector is not the project's only or primary claim. The current evidence is deliberately mixed: the two-stage model performs strongly on its locked MIT-BIH split, and two successive model generations have both shown a cross-database drop on INCART. That gap is preserved and documented rather than hidden.
10
+
11
+ ## Current evidence
12
+
13
+ **Two distinct model generations exist for the two-stage detector and must not be conflated** (see `validation_reports/VALIDATION_STATUS.md` for the full history). The original locked MIT-BIH row below is the `candidate-features-v3` model from 1.8.1. The current `candidate-features-v4` / windowed-std model now also has explicit MIT-BIH legacy non-regression audits, including the leakage-safe 0.0/0.0 polarity-gate result (F1 0.9644). Those audits do not become clean prospective evidence because MIT-BIH historically informed the adaptive-polarity mechanism.
14
+
15
+ | Protocol | Detector | Model generation | Records | Sensitivity | PPV | F1 |
16
+ |---|---|---|---:|---:|---:|---:|
17
+ | Locked MIT-BIH held-out | ElectroTrace two-stage | v3 (1.8.1 lock) | 12 | 0.9924 | 0.9879 | 0.9902 |
18
+ | MIT-BIH leakage-safe gate audit | ElectroTrace two-stage | v4 (legacy non-regression) | 12 | 0.9390 | 0.9913 | 0.9644 |
19
+ | Locked MIT-BIH held-out | Pan-Tompkins reimplementation | — | 12 | 0.9908 | 0.9954 | 0.9931 |
20
+ | Locked MIT-BIH held-out | Hamilton reimplementation | — | 12 | 0.9990 | 0.9303 | 0.9634 |
21
+ | Locked MIT-BIH held-out | ElectroTrace Stage-1 | — | 12 | 0.9931 | 0.7553 | 0.8580 |
22
+ | INCART external comparison | ElectroTrace two-stage | v3 | 68 | 0.3239 | 0.9679 | 0.4854 |
23
+ | INCART external comparison | ElectroTrace two-stage | v4 (windowed-std) | 68 | 0.9011 | 0.7594 | 0.8242 |
24
+ | INCART certified WFDB comparison | WFDB gqrs | — | 68 | 0.9324 | 0.9265 | 0.9294 |
25
+ | INCART complete source cohort | ElectroTrace two-stage | v4 frozen / development data | 75 | 0.8977 | 0.7633 | 0.8251 |
26
+ | INCART complete source cohort | WFDB gqrs | certified reference baseline | 75 | 0.9372 | 0.9310 | 0.9341 |
27
+ | INCART complete source cohort | WFDB sqrs | certified reference baseline | 75 | 0.7617 | 0.9537 | 0.8469 |
28
+ | European ST-T prospective external | ElectroTrace two-stage | v4 frozen | 90 | 0.9056 | 0.9644 | 0.9341 |
29
+
30
+ These values come from the repository's locked/archived validation artifacts. They are research results, not clinical validation and not evidence of universal detector superiority. **INCART informed v4 feature-scaling development and remains exposed development data**, so neither its historical 68-record row nor the completed 75-record row is a clean generalization estimate. The 75-record completion exists to finish the source cohort reproducibly, including the seven formerly malformed edge annotations under a two-certified-detector repair rule. A later one-shot prospective evaluation on all 90 European ST-T Database records provides independent external evidence for the unchanged frozen v4 configuration (F1 0.9341), but does not make INCART held-out or validate future post-hoc retuning. See `validation_reports/VALIDATION_STATUS.md` and `docs/CROSS_DATABASE_POLICY.md` for the evidence boundaries.
31
+
32
+ ## Why ElectroTrace exists
33
+
34
+ Most ECG software answers "what detector should I use?" ElectroTrace is aimed at a different question:
35
+
36
+ > **Under one explicit protocol, how does this detector behave, and can someone else reproduce the answer?**
37
+
38
+ Primary statistics should treat records or subjects as the experimental unit rather than pretending individual beats are independent biological replicates. Validation artifacts can carry dataset hashes, detector configuration, software version, git commit, and protocol metadata.
39
+
40
+ ### Primary use cases
41
+
42
+ **Preclinical electrophysiology:** batch animal recordings, retain explicit subject identifiers, and export record/subject-level tables for downstream statistics.
43
+
44
+ **Methods-heavy human ECG research:** evaluate an existing detector on held-out records and retain an auditable artifact for papers, reviews, and lab handoff.
45
+
46
+ **Detector development:** register a detector plugin and compare it against fixed baselines under the same matching rules.
47
+
48
+ ElectroTrace is research software. It is not a clinical device and the current models are not validated for clinical deployment.
49
+
50
+ ## Install
51
+
52
+ Python 3.10+ is required.
53
+
54
+ **v1.9.0 is release-ready, but this repository does not claim PyPI availability until the tagged Trusted Publishing workflow succeeds.** Until then, install from source:
55
+
56
+ ~~~bash
57
+ git clone https://github.com/Virelion-Biotech/Virelion-ElectroTrace.git
58
+ cd Virelion-ElectroTrace
59
+ python -m pip install -e ".[test,dev]"
60
+ ~~~
61
+
62
+ ## Five-minute start
63
+
64
+ ~~~bash
65
+ electrotrace list
66
+ electrotrace detect recording.edf --detector pan-tompkins --channel 0 -o peaks.csv
67
+ electrotrace batch data/ --detector pan-tompkins --workers 4 -o results/
68
+ electrotrace report validation.json -o validation.html
69
+ ~~~
70
+
71
+ Batch outputs:
72
+
73
+ ~~~text
74
+ results/
75
+ ├── beats.csv
76
+ ├── records.csv
77
+ ├── subjects.csv
78
+ ├── failures.csv
79
+ ├── manifest.json
80
+ ├── batch_state.json
81
+ └── peaks/
82
+ ~~~
83
+
84
+ Subject information is never inferred from filenames. Supply a two-column CSV with record and subject_id using --subject-map when subject aggregation is appropriate.
85
+
86
+ ## Validation and benchmarking
87
+
88
+ ~~~bash
89
+ electrotrace validate .cache/physionet/mitdb/100 --detector pan-tompkins -o validation.json
90
+ electrotrace bench .cache/physionet/mitdb --detectors pan-tompkins,hamilton --tolerance-ms 75 -o bench.json
91
+ ~~~
92
+
93
+ Bench currently operates on local WFDB records. The repository does not vendor PhysioNet data.
94
+
95
+ The frozen experimental QRS boundary locator has also completed a 105-record QT
96
+ Database tolerance-curve study. Reference-centered joint onset/offset success
97
+ was 0.589 at 20 ms, 0.845 at 40 ms, 0.925 at 60 ms, 0.954 at 80 ms, and 0.994
98
+ at 100 ms, with record-level bootstrap uncertainty. This is delineation
99
+ characterization, not an end-to-end detector or clinical-performance claim;
100
+ see `docs/QTDB_VALIDATION.md` and `docs/QRS_DELINEATION.md`.
101
+
102
+ The detector plugin interface is already present. The next benchmark phase will add external package adapters and a regenerated multi-database leaderboard rather than a single scalar ranking.
103
+
104
+ ## Model artifacts
105
+
106
+ The two-stage Random Forest is opt-in. New model persistence uses .skops plus a metadata sidecar. Legacy pickle models are migration-only and rejected by default.
107
+
108
+ A model records its feature schema and training scikit-learn major version. Loading stops when the runtime major version is incompatible or when the .skops file contains unknown serialized types.
109
+
110
+ The complete 75-record INCART result is part of the evidence surface: cross-database behavior must be measured instead of inferred from MIT-BIH. The research-use policy for unseen domains is documented in `docs/CROSS_DATABASE_POLICY.md`; ElectroTrace does not silently infer that a new database is in-domain or automatically promote the RF path over established reference baselines.
111
+
112
+ ## Validation philosophy
113
+
114
+ - record-level rather than beat-level primary statistics;
115
+ - one-to-one matching under a declared tolerance;
116
+ - locked splits with explicit seeds;
117
+ - bootstrap confidence intervals over records;
118
+ - SHA-256 input hashes and software/git provenance;
119
+ - separate timing error, sensitivity, PPV, and F1;
120
+ - explicit limitations and non-claims.
121
+
122
+ ## What is not claimed
123
+
124
+ ElectroTrace does not claim clinical performance, real-time streaming performance, population generalization, or superiority over established QRS detectors.
125
+
126
+ ## Documentation and citation
127
+
128
+ See docs/, mkdocs.yml, and CITATION.cff. The repository includes Zenodo metadata for v1.9.0, but a DOI is not claimed until an actual tagged release is archived by Zenodo.
129
+
130
+ ## License
131
+
132
+ AGPL-3.0-or-later at present. Any future core/server license split will be an explicit project decision.
@@ -0,0 +1,94 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "electrotrace"
7
+ version = "1.9.0"
8
+ description = "Reproducible ECG/electrophysiology annotation, benchmarking, phenotyping, and leakage-safe validation toolkit"
9
+ readme = "README.md"
10
+ license = { text = "AGPL-3.0-or-later" }
11
+ requires-python = ">=3.10"
12
+ classifiers = [
13
+ "Development Status :: 4 - Beta",
14
+ "Intended Audience :: Science/Research",
15
+ "License :: OSI Approved :: GNU Affero General Public License v3 or later (AGPLv3+)",
16
+ "Programming Language :: Python :: 3",
17
+ "Programming Language :: Python :: 3 :: Only",
18
+ "Programming Language :: Python :: 3.10",
19
+ "Programming Language :: Python :: 3.11",
20
+ "Programming Language :: Python :: 3.12",
21
+ "Programming Language :: Python :: 3.13",
22
+ "Topic :: Scientific/Engineering",
23
+ ]
24
+ dependencies = [
25
+ "numpy>=1.24",
26
+ "pandas>=2.0",
27
+ "scipy>=1.10",
28
+ "scikit-learn>=1.4",
29
+ ]
30
+
31
+ [project.optional-dependencies]
32
+ wfdb = ["wfdb>=4.1"]
33
+ edf = ["pyedflib>=0.1.38"]
34
+ server = ["flask>=3.0"]
35
+ models = ["skops>=0.14"]
36
+ all = [
37
+ "wfdb>=4.1",
38
+ "pyedflib>=0.1.38",
39
+ "flask>=3.0",
40
+ "skops>=0.14",
41
+ ]
42
+ test = [
43
+ "pytest>=8.0",
44
+ "wfdb>=4.1",
45
+ "pyedflib>=0.1.38",
46
+ "skops>=0.14",
47
+ ]
48
+ dev = [
49
+ "pytest>=8.0",
50
+ "ruff>=0.6",
51
+ "build>=1.2",
52
+ "mypy>=1.11",
53
+ "wfdb>=4.1",
54
+ "pyedflib>=0.1.38",
55
+ "flask>=3.0",
56
+ "skops>=0.14",
57
+ ]
58
+
59
+ [project.scripts]
60
+ electrotrace = "electrotrace.cli:main"
61
+ electrotrace-hearttwin = "electrotrace.hearttwin_adapter:main"
62
+
63
+ [project.entry-points."electrotrace.detectors"]
64
+ # Third-party detector packages register callables here.
65
+
66
+ [project.urls]
67
+ Homepage = "https://github.com/Virelion-Biotech/Virelion-ElectroTrace"
68
+ Documentation = "https://github.com/Virelion-Biotech/Virelion-ElectroTrace/tree/main/docs"
69
+ Repository = "https://github.com/Virelion-Biotech/Virelion-ElectroTrace"
70
+ Issues = "https://github.com/Virelion-Biotech/Virelion-ElectroTrace/issues"
71
+
72
+ [tool.setuptools.packages.find]
73
+ where = ["src"]
74
+
75
+ [tool.pytest.ini_options]
76
+ pythonpath = ["src", "."]
77
+ testpaths = ["tests"]
78
+ addopts = "--strict-markers"
79
+
80
+ [tool.ruff]
81
+ line-length = 120
82
+ target-version = "py310"
83
+
84
+ [tool.ruff.lint]
85
+ select = ["E", "F", "I", "UP"]
86
+
87
+ [tool.ruff.lint.isort]
88
+ known-first-party = ["electrotrace", "scripts"]
89
+
90
+ [tool.mypy]
91
+ python_version = "3.10"
92
+ ignore_missing_imports = true
93
+ check_untyped_defs = true
94
+ warn_unused_ignores = true
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,28 @@
1
+ """ElectroTrace: reproducible ECG/electrophysiology research and benchmarking tools."""
2
+
3
+ __version__ = "1.9.0"
4
+
5
+ from .candidate_suppressor import CandidateSuppressor
6
+ from .io import load_recording
7
+ from .provenance import DatasetManifest, manifest_from_dict
8
+ from .qrs_delineation import delineate_qrs
9
+ from .research_validation import build_validation_report, summarize_records_rigorous, write_validation_report
10
+ from .signal import apply_pipeline
11
+ from .validation import validate_record
12
+ from .validation_detectors import detect_r_peaks, detect_r_peaks_two_stage
13
+
14
+ __all__ = [
15
+ "__version__",
16
+ "CandidateSuppressor",
17
+ "DatasetManifest",
18
+ "manifest_from_dict",
19
+ "load_recording",
20
+ "apply_pipeline",
21
+ "detect_r_peaks",
22
+ "detect_r_peaks_two_stage",
23
+ "delineate_qrs",
24
+ "validate_record",
25
+ "build_validation_report",
26
+ "summarize_records_rigorous",
27
+ "write_validation_report",
28
+ ]
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ raise SystemExit(main())
@@ -0,0 +1,194 @@
1
+ """Annotation model, validation, serialization, review state, and agreement metrics."""
2
+ from __future__ import annotations
3
+
4
+ import json
5
+ import math
6
+ import uuid
7
+ from dataclasses import asdict, dataclass, field
8
+ from typing import Literal, Optional
9
+
10
+ AnnotationType = Literal["interval", "point"]
11
+ ReviewStatus = Literal["unreviewed", "accepted", "flagged"]
12
+ DEFAULT_LABELS = ["P Wave", "QRS", "T Wave", "Pacemaker Spike", "R Peak", "Artifact", "Abnormal Beat", "Graft Activation", "Arrhythmia", "Custom"]
13
+
14
+
15
+ @dataclass
16
+ class Annotation:
17
+ label: str
18
+ type: AnnotationType
19
+ channel: str
20
+ start: Optional[float] = None
21
+ end: Optional[float] = None
22
+ time: Optional[float] = None
23
+ confidence: float = 1.0
24
+ notes: str = ""
25
+ annotator: str = ""
26
+ status: ReviewStatus = "unreviewed"
27
+ reviewer: str = ""
28
+ review_notes: str = ""
29
+ id: str = field(default_factory=lambda: uuid.uuid4().hex[:10])
30
+
31
+ def validate(self, duration_s: float | None = None, start_time_s: float = 0.0, end_time_s: float | None = None) -> None:
32
+ if not self.label.strip():
33
+ raise ValueError("label must not be empty")
34
+ if not self.channel.strip():
35
+ raise ValueError("channel must not be empty")
36
+ if not math.isfinite(float(self.confidence)) or not 0.0 <= self.confidence <= 1.0:
37
+ raise ValueError("confidence must be between 0 and 1")
38
+ if self.status not in {"unreviewed", "accepted", "flagged"}:
39
+ raise ValueError("invalid review status")
40
+ start_bound = float(start_time_s)
41
+ end_bound = float(end_time_s) if end_time_s is not None else (start_bound + float(duration_s) if duration_s is not None else None)
42
+ if not math.isfinite(start_bound) or (end_bound is not None and not math.isfinite(end_bound)) or (end_bound is not None and end_bound < start_bound):
43
+ raise ValueError("invalid recording time bounds")
44
+ if self.type == "interval":
45
+ if self.start is None or self.end is None:
46
+ raise ValueError("interval annotations require start and end")
47
+ if not math.isfinite(self.start) or not math.isfinite(self.end):
48
+ raise ValueError("interval coordinates must be finite")
49
+ if self.end <= self.start:
50
+ raise ValueError("end must be greater than start")
51
+ if self.time is not None:
52
+ raise ValueError("interval annotations must not define time")
53
+ if end_bound is not None and (self.start < start_bound or self.end > end_bound):
54
+ raise ValueError("interval is outside the recording bounds")
55
+ elif self.type == "point":
56
+ if self.time is None:
57
+ raise ValueError("point annotations require time")
58
+ if not math.isfinite(self.time):
59
+ raise ValueError("point time must be finite")
60
+ if self.start is not None or self.end is not None:
61
+ raise ValueError("point annotations must not define start/end")
62
+ if end_bound is not None and not start_bound <= self.time <= end_bound:
63
+ raise ValueError("point is outside the recording bounds")
64
+ else:
65
+ raise ValueError(f"unknown annotation type: {self.type}")
66
+
67
+ @property
68
+ def position(self) -> float:
69
+ return self.start if self.type == "interval" else self.time # type: ignore[return-value]
70
+
71
+ def to_dict(self) -> dict:
72
+ return asdict(self)
73
+
74
+
75
+ class AnnotationStore:
76
+ def __init__(self, duration_s: float | None = None, start_time_s: float = 0.0, end_time_s: float | None = None):
77
+ self.duration_s = duration_s
78
+ self.start_time_s = float(start_time_s)
79
+ self.end_time_s = float(end_time_s) if end_time_s is not None else (self.start_time_s + float(duration_s) if duration_s is not None else None)
80
+ self._items: list[Annotation] = []
81
+
82
+ @property
83
+ def items(self) -> list[Annotation]:
84
+ return list(self._items)
85
+
86
+ def add(self, ann: Annotation) -> Annotation:
87
+ ann.validate(self.duration_s, self.start_time_s, self.end_time_s)
88
+ if any(a.id == ann.id for a in self._items):
89
+ raise ValueError(f"duplicate annotation id: {ann.id}")
90
+ self._items.append(ann)
91
+ self._sort()
92
+ return ann
93
+
94
+ def update(self, ann_id: str, **changes) -> Annotation:
95
+ for idx, existing in enumerate(self._items):
96
+ if existing.id == ann_id:
97
+ data = {**asdict(existing), **changes}
98
+ updated = Annotation(**data)
99
+ updated.validate(self.duration_s, self.start_time_s, self.end_time_s)
100
+ self._items[idx] = updated
101
+ self._sort()
102
+ return updated
103
+ raise KeyError(ann_id)
104
+
105
+ def delete(self, ann_id: str) -> bool:
106
+ before = len(self._items)
107
+ self._items = [a for a in self._items if a.id != ann_id]
108
+ return len(self._items) != before
109
+
110
+ def duplicate(self, ann_id: str) -> Annotation:
111
+ for existing in self._items:
112
+ if existing.id == ann_id:
113
+ data = asdict(existing)
114
+ data["id"] = uuid.uuid4().hex[:10]
115
+ data["status"] = "unreviewed"
116
+ data["reviewer"] = ""
117
+ data["review_notes"] = ""
118
+ return self.add(Annotation(**data))
119
+ raise KeyError(ann_id)
120
+
121
+ def clear(self) -> None:
122
+ self._items.clear()
123
+
124
+ def _sort(self) -> None:
125
+ self._items.sort(key=lambda a: a.position)
126
+
127
+ def to_dict(self, source_file: str = "", metadata: dict | None = None) -> dict:
128
+ meta = dict(metadata or {})
129
+ meta.setdefault("time_start_s", self.start_time_s)
130
+ if self.end_time_s is not None:
131
+ meta.setdefault("time_end_s", self.end_time_s)
132
+ return {"schema": "electrotrace.annotation/v2", "file": source_file, "metadata": meta, "annotations": [a.to_dict() for a in self._items]}
133
+
134
+ def to_json(self, source_file: str = "", metadata: dict | None = None) -> str:
135
+ return json.dumps(self.to_dict(source_file, metadata), indent=2)
136
+
137
+ @classmethod
138
+ def from_dict(cls, data: dict, duration_s: float | None = None, start_time_s: float | None = None, end_time_s: float | None = None) -> "AnnotationStore":
139
+ metadata = data.get("metadata") or {}
140
+ start = float(metadata.get("time_start_s", 0.0) if start_time_s is None else start_time_s)
141
+ end = metadata.get("time_end_s") if end_time_s is None else end_time_s
142
+ store = cls(duration_s=duration_s, start_time_s=start, end_time_s=float(end) if end is not None else None)
143
+ schema = data.get("schema", data.get("annotation_schema", ""))
144
+ if schema and schema not in {"electrotrace.annotation/v2", "v1"}:
145
+ raise ValueError(f"unsupported annotation schema: {schema}")
146
+ for raw in data.get("annotations", []):
147
+ store.add(Annotation(**raw))
148
+ return store
149
+
150
+ @classmethod
151
+ def from_json(cls, text: str, duration_s: float | None = None, start_time_s: float | None = None, end_time_s: float | None = None) -> "AnnotationStore":
152
+ try:
153
+ data = json.loads(text)
154
+ except json.JSONDecodeError as exc:
155
+ raise ValueError(f"invalid JSON: {exc}") from exc
156
+ return cls.from_dict(data, duration_s=duration_s, start_time_s=start_time_s, end_time_s=end_time_s)
157
+
158
+ def to_csv_rows(self, source_file: str = "") -> list[dict]:
159
+ return [{"file": source_file, "id": a.id, "type": a.type, "label": a.label, "channel": a.channel, "start": a.start, "end": a.end, "time": a.time, "confidence": a.confidence, "notes": a.notes, "annotator": a.annotator, "status": a.status, "reviewer": a.reviewer, "review_notes": a.review_notes} for a in self._items]
160
+
161
+
162
+ def point_agreement(a: list[Annotation], b: list[Annotation], tolerance_s: float = 0.04) -> dict:
163
+ if tolerance_s <= 0 or not math.isfinite(tolerance_s):
164
+ raise ValueError("tolerance_s must be positive and finite")
165
+ left = [x for x in a if x.type == "point"]
166
+ right = [x for x in b if x.type == "point"]
167
+ used: set[str] = set()
168
+ errors = []
169
+ matches = 0
170
+ for x in left:
171
+ candidates = [
172
+ y for y in right
173
+ if y.id not in used
174
+ and y.label == x.label
175
+ and y.channel == x.channel
176
+ and y.time is not None
177
+ and x.time is not None
178
+ and abs(y.time - x.time) <= tolerance_s
179
+ ]
180
+ if candidates:
181
+ y = min(candidates, key=lambda z: abs(z.time - x.time))
182
+ used.add(y.id)
183
+ matches += 1
184
+ errors.append(abs(y.time - x.time))
185
+ total = max(len(left), len(right), 1)
186
+ return {"matches": matches, "agreement_rate": matches / total, "mean_absolute_error_s": sum(errors) / len(errors) if errors else None}
187
+
188
+
189
+ def interval_iou(a: Annotation, b: Annotation) -> float:
190
+ if a.type != "interval" or b.type != "interval":
191
+ return 0.0
192
+ inter = max(0.0, min(a.end, b.end) - max(a.start, b.start)) # type: ignore[arg-type]
193
+ union = max(a.end, b.end) - min(a.start, b.start) # type: ignore[arg-type]
194
+ return inter / union if union else 0.0