wellerlab 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- wellerlab-0.1.0/PKG-INFO +77 -0
- wellerlab-0.1.0/README.md +62 -0
- wellerlab-0.1.0/setup.cfg +4 -0
- wellerlab-0.1.0/setup.py +54 -0
- wellerlab-0.1.0/wellerlab/__init__.py +17 -0
- wellerlab-0.1.0/wellerlab/metabo/__init__.py +3 -0
- wellerlab-0.1.0/wellerlab/metabo/metabo_core.py +282 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/__init__.py +17 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/_dialogs.py +37 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/_plot.py +194 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/icons/FeatureFilter.svg +6 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/icons/FeatureImport.svg +6 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/icons/Heatmap.svg +6 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/icons/Preprocess.svg +6 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/icons/UnivariateStats.svg +6 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/icons/Volcano.svg +6 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/owfeaturefilter.py +137 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/owfeatureimport.py +118 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/owheatmap.py +267 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/owpreprocess.py +164 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/owunivariate.py +185 -0
- wellerlab-0.1.0/wellerlab/metabo/widgets/owvolcano.py +512 -0
- wellerlab-0.1.0/wellerlab/nmr/__init__.py +14 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/__init__.py +17 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRBaseline.svg +6 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRBinning.svg +6 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRExclude.svg +6 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRFilter.svg +6 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRNormalize.svg +6 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRReference.svg +6 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/nmr_utils.py +74 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrbaseline.py +149 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrbinning.py +124 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrexclude.py +145 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrfilter.py +126 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrnormalize.py +134 -0
- wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrreference.py +145 -0
- wellerlab-0.1.0/wellerlab/pca/__init__.py +16 -0
- wellerlab-0.1.0/wellerlab/pca/pca_analysis.py +246 -0
- wellerlab-0.1.0/wellerlab/pca/widgets/__init__.py +7 -0
- wellerlab-0.1.0/wellerlab/pca/widgets/icons/PCAPro.svg +6 -0
- wellerlab-0.1.0/wellerlab/pca/widgets/owpcawell.py +563 -0
- wellerlab-0.1.0/wellerlab/plsda/__init__.py +29 -0
- wellerlab-0.1.0/wellerlab/plsda/opls_core.py +377 -0
- wellerlab-0.1.0/wellerlab/plsda/oplsda_learner.py +209 -0
- wellerlab-0.1.0/wellerlab/plsda/plsda_learner.py +120 -0
- wellerlab-0.1.0/wellerlab/plsda/widgets/__init__.py +8 -0
- wellerlab-0.1.0/wellerlab/plsda/widgets/icons/OPLSDA.svg +6 -0
- wellerlab-0.1.0/wellerlab/plsda/widgets/icons/PLSDA.svg +6 -0
- wellerlab-0.1.0/wellerlab/plsda/widgets/owoplsda.py +506 -0
- wellerlab-0.1.0/wellerlab/plsda/widgets/owplsda.py +192 -0
- wellerlab-0.1.0/wellerlab/widgets/__init__.py +17 -0
- wellerlab-0.1.0/wellerlab.egg-info/PKG-INFO +77 -0
- wellerlab-0.1.0/wellerlab.egg-info/SOURCES.txt +56 -0
- wellerlab-0.1.0/wellerlab.egg-info/dependency_links.txt +1 -0
- wellerlab-0.1.0/wellerlab.egg-info/entry_points.txt +2 -0
- wellerlab-0.1.0/wellerlab.egg-info/requires.txt +6 -0
- wellerlab-0.1.0/wellerlab.egg-info/top_level.txt +1 -0
wellerlab-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: wellerlab
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: WellerLab Orange3 suite: Metabo statistics, PLS-DA/OPLS-DA, PCA Pro and NMR preprocessing in one category
|
|
5
|
+
Author: Philipp Weller
|
|
6
|
+
Author-email: philipp.weller@googlemail.com
|
|
7
|
+
Project-URL: Source, https://github.com/philippweller/WellerLab/tree/main/orange-wellerlab-addon
|
|
8
|
+
Project-URL: Bug Tracker, https://github.com/philippweller/WellerLab/issues
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
|
|
16
|
+
# wellerlab — WellerLab Orange3 Suitepack
|
|
17
|
+
|
|
18
|
+
Alle Werkzeuge des Weller-Labors in **einem** Orange3-Add-on und **einer**
|
|
19
|
+
Kategorie: **Weller Lab**.
|
|
20
|
+
|
|
21
|
+
| Familie | Widgets | Inhalt |
|
|
22
|
+
|---|---|---|
|
|
23
|
+
| `wellerlab.metabo` | Metabo Feature Table, Preprocess, Feature Filter, Univariate Stats, Heatmap, Volcano | MetaboAnalyst-artige Statistik für GC-MS/GC-IMS-Feature-Tables (Compound Discoverer). Qt-freier Kern `metabo_core`. |
|
|
24
|
+
| `wellerlab.plsda` | PLS-DA, OPLS-DA | Multivariate Klassifikation; OPLS-DA mit S-Plot (p1/p(corr)), Schwellen, Farbcodierung, Klick/Shift-Klick-Selektion, VIP + orthoVIP, R2X/R2Y/Q2 (7-fach-CV) und Permutationstest. Qt-freier Kern `opls_core`. |
|
|
25
|
+
| `wellerlab.pca` | PCA Pro | Chemometrische PCA: Scaling (none/center/Pareto/autoscale), Komponentenwahl (Kaiser, Varianzanteil, feste Zahl), Hotelling-T²- und Q-Residuen-Diagnostik. |
|
|
26
|
+
| `wellerlab.nmr` | NMR Baseline Correction, Binning (Bucketing), Region Exclusion, Filter (Savitzky-Golay), Normalization, Reference & Alignment | Vorverarbeitung von NMR-Spektren. |
|
|
27
|
+
|
|
28
|
+
## Installation
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
# aus dem Monorepo (empfohlen): findet Oranges eigenes Python
|
|
32
|
+
python orange-install.py wellerlab
|
|
33
|
+
|
|
34
|
+
# oder direkt
|
|
35
|
+
/Applications/Orange.app/Contents/MacOS/python -m pip install wellerlab
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Beim ersten Start danach erscheint die Kategorie **Weller Lab** mit allen
|
|
39
|
+
Widgets. Wichtig: frühere Einzelpakete (`orangemetabo`, `orangeplsda`,
|
|
40
|
+
`orangepca`, `orangenmr`) müssen entfernt sein, sonst erscheinen die Widgets
|
|
41
|
+
doppelt.
|
|
42
|
+
|
|
43
|
+
## Icons
|
|
44
|
+
|
|
45
|
+
Ein generiertes, konsistentes Set: gleiche Kachelform und Strichstärke,
|
|
46
|
+
Familienfarbe und -glyphe (Chromatogramm = Metabo, Multiplett = NMR,
|
|
47
|
+
latente Ellipsen = PLS/OPLS, Score-Plot = PCA) plus Kurzlabel.
|
|
48
|
+
Erzeugt von `tools/gen_icons.py` im Monorepo.
|
|
49
|
+
|
|
50
|
+
## Tests (headless, offscreen)
|
|
51
|
+
|
|
52
|
+
Kanonischer Befehl — findet Oranges Python selbst und läuft mit *jedem* Python:
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
python3 run_tests.py # alle Suiten
|
|
56
|
+
python3 run_tests.py --python /pfad/zum/python
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Die drei Suiten einzeln:
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
PY=/Applications/Orange.app/Contents/MacOS/python
|
|
63
|
+
$PY _test_opls_core.py # Numerik: R2Y/Q2/VIP/Permutation
|
|
64
|
+
$PY _test_owoplsda.py # Widget: Schwellen, Farben, Selektion, Ausgänge
|
|
65
|
+
$PY _test_suite.py # Integration: alle 15 Widgets, Icons, Kategorie
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
> **macOS/Apple-Silicon-Falle:** Wird Oranges universeller Interpreter von einem
|
|
69
|
+
> x86_64-Python (Rosetta, z. B. einer Intel-conda-Installation) gestartet, läuft er
|
|
70
|
+
> selbst als x86_64 — dort scheitert der Import von Oranges arm64-only
|
|
71
|
+
> numpy-Extension („you should not try to import numpy from its source
|
|
72
|
+
> directory“). `run_tests.py` umgeht das automatisch mit `arch -arm64`; bei
|
|
73
|
+
> direkten Aufrufen nativ starten oder ebenfalls `arch -arm64` voranstellen.
|
|
74
|
+
|
|
75
|
+
## Lizenz
|
|
76
|
+
|
|
77
|
+
MIT · Philipp Weller · philipp.weller@googlemail.com
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
# wellerlab — WellerLab Orange3 Suitepack
|
|
2
|
+
|
|
3
|
+
Alle Werkzeuge des Weller-Labors in **einem** Orange3-Add-on und **einer**
|
|
4
|
+
Kategorie: **Weller Lab**.
|
|
5
|
+
|
|
6
|
+
| Familie | Widgets | Inhalt |
|
|
7
|
+
|---|---|---|
|
|
8
|
+
| `wellerlab.metabo` | Metabo Feature Table, Preprocess, Feature Filter, Univariate Stats, Heatmap, Volcano | MetaboAnalyst-artige Statistik für GC-MS/GC-IMS-Feature-Tables (Compound Discoverer). Qt-freier Kern `metabo_core`. |
|
|
9
|
+
| `wellerlab.plsda` | PLS-DA, OPLS-DA | Multivariate Klassifikation; OPLS-DA mit S-Plot (p1/p(corr)), Schwellen, Farbcodierung, Klick/Shift-Klick-Selektion, VIP + orthoVIP, R2X/R2Y/Q2 (7-fach-CV) und Permutationstest. Qt-freier Kern `opls_core`. |
|
|
10
|
+
| `wellerlab.pca` | PCA Pro | Chemometrische PCA: Scaling (none/center/Pareto/autoscale), Komponentenwahl (Kaiser, Varianzanteil, feste Zahl), Hotelling-T²- und Q-Residuen-Diagnostik. |
|
|
11
|
+
| `wellerlab.nmr` | NMR Baseline Correction, Binning (Bucketing), Region Exclusion, Filter (Savitzky-Golay), Normalization, Reference & Alignment | Vorverarbeitung von NMR-Spektren. |
|
|
12
|
+
|
|
13
|
+
## Installation
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
# aus dem Monorepo (empfohlen): findet Oranges eigenes Python
|
|
17
|
+
python orange-install.py wellerlab
|
|
18
|
+
|
|
19
|
+
# oder direkt
|
|
20
|
+
/Applications/Orange.app/Contents/MacOS/python -m pip install wellerlab
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Beim ersten Start danach erscheint die Kategorie **Weller Lab** mit allen
|
|
24
|
+
Widgets. Wichtig: frühere Einzelpakete (`orangemetabo`, `orangeplsda`,
|
|
25
|
+
`orangepca`, `orangenmr`) müssen entfernt sein, sonst erscheinen die Widgets
|
|
26
|
+
doppelt.
|
|
27
|
+
|
|
28
|
+
## Icons
|
|
29
|
+
|
|
30
|
+
Ein generiertes, konsistentes Set: gleiche Kachelform und Strichstärke,
|
|
31
|
+
Familienfarbe und -glyphe (Chromatogramm = Metabo, Multiplett = NMR,
|
|
32
|
+
latente Ellipsen = PLS/OPLS, Score-Plot = PCA) plus Kurzlabel.
|
|
33
|
+
Erzeugt von `tools/gen_icons.py` im Monorepo.
|
|
34
|
+
|
|
35
|
+
## Tests (headless, offscreen)
|
|
36
|
+
|
|
37
|
+
Kanonischer Befehl — findet Oranges Python selbst und läuft mit *jedem* Python:
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
python3 run_tests.py # alle Suiten
|
|
41
|
+
python3 run_tests.py --python /pfad/zum/python
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Die drei Suiten einzeln:
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
PY=/Applications/Orange.app/Contents/MacOS/python
|
|
48
|
+
$PY _test_opls_core.py # Numerik: R2Y/Q2/VIP/Permutation
|
|
49
|
+
$PY _test_owoplsda.py # Widget: Schwellen, Farben, Selektion, Ausgänge
|
|
50
|
+
$PY _test_suite.py # Integration: alle 15 Widgets, Icons, Kategorie
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
> **macOS/Apple-Silicon-Falle:** Wird Oranges universeller Interpreter von einem
|
|
54
|
+
> x86_64-Python (Rosetta, z. B. einer Intel-conda-Installation) gestartet, läuft er
|
|
55
|
+
> selbst als x86_64 — dort scheitert der Import von Oranges arm64-only
|
|
56
|
+
> numpy-Extension („you should not try to import numpy from its source
|
|
57
|
+
> directory“). `run_tests.py` umgeht das automatisch mit `arch -arm64`; bei
|
|
58
|
+
> direkten Aufrufen nativ starten oder ebenfalls `arch -arm64` voranstellen.
|
|
59
|
+
|
|
60
|
+
## Lizenz
|
|
61
|
+
|
|
62
|
+
MIT · Philipp Weller · philipp.weller@googlemail.com
|
wellerlab-0.1.0/setup.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
from setuptools import setup, find_packages
|
|
4
|
+
|
|
5
|
+
HERE = Path(__file__).parent
|
|
6
|
+
|
|
7
|
+
README = HERE / "README.md"
|
|
8
|
+
long_description = (README.read_text(encoding="utf-8") if README.exists()
|
|
9
|
+
else "WellerLab Orange3 tool suite: metabolomics feature-table "
|
|
10
|
+
"statistics, PLS-DA/OPLS-DA, PCA Pro and NMR preprocessing.")
|
|
11
|
+
|
|
12
|
+
setup(
|
|
13
|
+
name="wellerlab",
|
|
14
|
+
version="0.1.0",
|
|
15
|
+
description="WellerLab Orange3 suite: Metabo statistics, PLS-DA/OPLS-DA, "
|
|
16
|
+
"PCA Pro and NMR preprocessing in one category",
|
|
17
|
+
long_description=long_description,
|
|
18
|
+
long_description_content_type="text/markdown",
|
|
19
|
+
packages=find_packages(include=["wellerlab", "wellerlab.*"]),
|
|
20
|
+
include_package_data=True,
|
|
21
|
+
author="Philipp Weller",
|
|
22
|
+
author_email="philipp.weller@googlemail.com",
|
|
23
|
+
install_requires=[
|
|
24
|
+
"Orange3>=3.40.0",
|
|
25
|
+
"numpy",
|
|
26
|
+
"scipy",
|
|
27
|
+
"matplotlib",
|
|
28
|
+
"scikit-learn",
|
|
29
|
+
"pandas",
|
|
30
|
+
],
|
|
31
|
+
entry_points={
|
|
32
|
+
"orange.widgets": (
|
|
33
|
+
"Weller Lab = wellerlab.widgets",
|
|
34
|
+
),
|
|
35
|
+
},
|
|
36
|
+
package_data={
|
|
37
|
+
"wellerlab": ["widgets/icons/*.svg",
|
|
38
|
+
"metabo/widgets/icons/*.svg",
|
|
39
|
+
"plsda/widgets/icons/*.svg",
|
|
40
|
+
"pca/widgets/icons/*.svg",
|
|
41
|
+
"nmr/widgets/icons/*.svg"],
|
|
42
|
+
},
|
|
43
|
+
project_urls={
|
|
44
|
+
"Source": "https://github.com/philippweller/WellerLab/tree/main/orange-wellerlab-addon",
|
|
45
|
+
"Bug Tracker": "https://github.com/philippweller/WellerLab/issues",
|
|
46
|
+
},
|
|
47
|
+
classifiers=[
|
|
48
|
+
"Development Status :: 3 - Alpha",
|
|
49
|
+
"Intended Audience :: Science/Research",
|
|
50
|
+
"License :: OSI Approved :: MIT License",
|
|
51
|
+
"Programming Language :: Python :: 3",
|
|
52
|
+
"Topic :: Scientific/Engineering",
|
|
53
|
+
],
|
|
54
|
+
)
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""
|
|
2
|
+
WellerLab — the complete Orange3 tool suite of the Weller lab.
|
|
3
|
+
|
|
4
|
+
One distribution, one Orange category ("Weller Lab"):
|
|
5
|
+
|
|
6
|
+
wellerlab.metabo MetaboAnalyst-style feature-table statistics
|
|
7
|
+
(import, preprocess, filter, univariate, heatmap, volcano)
|
|
8
|
+
wellerlab.plsda PLS-DA and OPLS-DA (S-plot, VIP, Q2, permutation)
|
|
9
|
+
wellerlab.pca PCA Pro
|
|
10
|
+
wellerlab.nmr NMR preprocessing (baseline, binning, exclusion,
|
|
11
|
+
filter, normalization, reference)
|
|
12
|
+
|
|
13
|
+
The numerical cores (``metabo.metabo_core``, ``plsda.opls_core``) are free of
|
|
14
|
+
Qt/Orange and can be imported and tested headlessly.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
"""
|
|
4
|
+
MetaboAnalyst-style univariate statistics core for GC-MS / GC-IMS
|
|
5
|
+
feature tables (Compound Discoverer exports).
|
|
6
|
+
|
|
7
|
+
This module is the analytics heart of the wellerlab.metabo add-on. It is
|
|
8
|
+
deliberately free of any Qt / Orange dependency so it can be unit-tested
|
|
9
|
+
headlessly with plain numpy/scipy. The widgets in `orangemetabo/widgets/`
|
|
10
|
+
are thin GUI wrappers around the functions here.
|
|
11
|
+
|
|
12
|
+
Pipeline (matches the validated ground truth `cv_anova_alle_97_features.csv`,
|
|
13
|
+
97/97 features on F, p and FDR_BH):
|
|
14
|
+
|
|
15
|
+
load -> sum-normalise -> log2 -> autoscale (z-score per feature)
|
|
16
|
+
-> univariate statistics (one-way ANOVA / Welch t / Kruskal-Wallis)
|
|
17
|
+
-> Benjamini-Hochberg FDR
|
|
18
|
+
|
|
19
|
+
All functions are pure: they take arrays / lists and return DataFrames or
|
|
20
|
+
arrays. No file I/O except the optional `load_feature_table` reader.
|
|
21
|
+
"""
|
|
22
|
+
import csv
|
|
23
|
+
import numpy as np
|
|
24
|
+
import pandas as pd
|
|
25
|
+
from scipy import stats
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# --------------------------------------------------------------------------
|
|
29
|
+
# Feature-table reader (Compound Discoverer 2-header semicolon CSV)
|
|
30
|
+
# --------------------------------------------------------------------------
|
|
31
|
+
|
|
32
|
+
def load_feature_table(path):
|
|
33
|
+
"""Read a Compound Discoverer feature-table CSV.
|
|
34
|
+
|
|
35
|
+
Expected layout (semicolon-separated, UTF-8-BOM):
|
|
36
|
+
row 0: "" , "Area: <sample>.raw", ...
|
|
37
|
+
row 1: "" , "<group>", ...
|
|
38
|
+
row 2+: "<feature name>", values...
|
|
39
|
+
|
|
40
|
+
Returns (feature_names: list[str], samples: list[str],
|
|
41
|
+
groups: list[str], X: np.ndarray[features x samples]).
|
|
42
|
+
"""
|
|
43
|
+
with open(path, encoding="utf-8-sig", newline="") as f:
|
|
44
|
+
rows = list(csv.reader(f, delimiter=";"))
|
|
45
|
+
samples = [h.replace("Area: ", "").replace(".raw", "").strip()
|
|
46
|
+
for h in rows[0][1:]]
|
|
47
|
+
groups = [g.strip() for g in rows[1][1:]]
|
|
48
|
+
feat = [r[0].strip() for r in rows[2:]]
|
|
49
|
+
X = np.array(
|
|
50
|
+
[[float(v) for v in r[1:1 + len(samples)]] for r in rows[2:]],
|
|
51
|
+
dtype=float,
|
|
52
|
+
)
|
|
53
|
+
return feat, samples, groups, X
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
# --------------------------------------------------------------------------
|
|
57
|
+
# Preprocessing
|
|
58
|
+
# --------------------------------------------------------------------------
|
|
59
|
+
|
|
60
|
+
def normalize_sum(X):
|
|
61
|
+
"""Total-area (sum) normalisation: rescale each sample column so that
|
|
62
|
+
every column has the mean of the column totals."""
|
|
63
|
+
col_sums = X.sum(axis=0)
|
|
64
|
+
return X / col_sums * col_sums.mean()
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def log2_transform(X):
|
|
68
|
+
"""log2 with a guard against non-positive values (clip to tiny eps)."""
|
|
69
|
+
eps = np.finfo(float).tiny
|
|
70
|
+
return np.log2(np.clip(X, eps, None))
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def impute_low(X, threshold=0.05, method="knn", k=3):
|
|
74
|
+
"""Impute below-threshold values (fraction of column min or a hard cap).
|
|
75
|
+
|
|
76
|
+
Methods:
|
|
77
|
+
'constant' : replace with the column minimum (or 0).
|
|
78
|
+
'min' : replace with the column minimum.
|
|
79
|
+
'knn' : replace with the mean of the k nearest samples in the
|
|
80
|
+
Euclidean distance of the non-imputed rows.
|
|
81
|
+
Returns (X_imputed, imputed_mask).
|
|
82
|
+
"""
|
|
83
|
+
X = np.asarray(X, dtype=float).copy()
|
|
84
|
+
col_min = X.min(axis=0)
|
|
85
|
+
mask = X < (col_min * (1 + threshold))
|
|
86
|
+
if not mask.any():
|
|
87
|
+
return X, mask
|
|
88
|
+
if method in ("constant", "min"):
|
|
89
|
+
rows, cols = np.nonzero(mask)
|
|
90
|
+
X[rows, cols] = col_min[cols]
|
|
91
|
+
return X, mask
|
|
92
|
+
# knn: distance between samples (columns) over all feature rows;
|
|
93
|
+
# impute each flagged value with the mean of its k nearest samples.
|
|
94
|
+
n_rows, n_cols = X.shape
|
|
95
|
+
d = np.sqrt(((X[:, :, None] - X[:, None, :]) ** 2).sum(axis=0)) # c x c
|
|
96
|
+
np.fill_diagonal(d, np.inf)
|
|
97
|
+
order = d.argsort(axis=1)
|
|
98
|
+
for c in range(n_cols):
|
|
99
|
+
bad = np.nonzero(mask[:, c])[0]
|
|
100
|
+
if bad.size == 0:
|
|
101
|
+
continue
|
|
102
|
+
for r in bad:
|
|
103
|
+
# prefer samples that are not flagged on this same feature
|
|
104
|
+
nb = order[c, :max(k, 1)]
|
|
105
|
+
ok = [j for j in nb if not (mask[r, j] and j != c)]
|
|
106
|
+
if not ok:
|
|
107
|
+
X[r, c] = col_min[c]
|
|
108
|
+
else:
|
|
109
|
+
X[r, c] = X[r, ok].mean()
|
|
110
|
+
return X, mask
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def scale_rows(X, method="autoscale"):
|
|
114
|
+
"""Per-feature (row) scaling.
|
|
115
|
+
|
|
116
|
+
'autoscale' : z-score (mean 0, std 1, ddof=1)
|
|
117
|
+
'pareto' : (x - mean) / sqrt(std)
|
|
118
|
+
'none' : unchanged
|
|
119
|
+
Constant rows are returned as zeros (division guard).
|
|
120
|
+
"""
|
|
121
|
+
X = np.asarray(X, dtype=float)
|
|
122
|
+
mean = X.mean(axis=1, keepdims=True)
|
|
123
|
+
if method == "none":
|
|
124
|
+
return X
|
|
125
|
+
std = X.std(axis=1, keepdims=True, ddof=1)
|
|
126
|
+
denom = np.where(std > 0, np.sqrt(std) if method == "pareto" else std, 1.0)
|
|
127
|
+
return (X - mean) / denom
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
# --------------------------------------------------------------------------
|
|
131
|
+
# Multiple-testing correction
|
|
132
|
+
# --------------------------------------------------------------------------
|
|
133
|
+
|
|
134
|
+
def bh_fdr(p):
|
|
135
|
+
"""Benjamini-Hochberg FDR for an array of p-values."""
|
|
136
|
+
p = np.asarray(p, dtype=float)
|
|
137
|
+
m = len(p)
|
|
138
|
+
if m == 0:
|
|
139
|
+
return p
|
|
140
|
+
order = p.argsort()
|
|
141
|
+
ranked = p[order]
|
|
142
|
+
q = np.minimum.accumulate((ranked * m / np.arange(1, m + 1))[::-1])[::-1]
|
|
143
|
+
qf = np.empty(m)
|
|
144
|
+
qf[order] = np.clip(q, 0, 1)
|
|
145
|
+
return qf
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
# --------------------------------------------------------------------------
|
|
149
|
+
# Univariate statistics
|
|
150
|
+
# --------------------------------------------------------------------------
|
|
151
|
+
|
|
152
|
+
def _groups_arrays(X, groups, levels):
|
|
153
|
+
"""Yield the per-group column slices for every feature row."""
|
|
154
|
+
gi = {lv: np.array(groups) == lv for lv in levels}
|
|
155
|
+
return [X[:, idx] for idx in gi.values()]
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def univariate(X, groups, method="anova", base=None, treats=None,
|
|
159
|
+
feature_names=None):
|
|
160
|
+
"""Run the selected univariate test for every feature row.
|
|
161
|
+
|
|
162
|
+
method:
|
|
163
|
+
'anova' : one-way ANOVA across all levels (>= 2 levels, >= 2 per level).
|
|
164
|
+
'welch' : Welch two-sample t-test, `treats` combined vs `base`.
|
|
165
|
+
'kruskal': one-way Kruskal-Wallis across all levels.
|
|
166
|
+
|
|
167
|
+
feature_names: optional list of row labels (same length as X rows);
|
|
168
|
+
when given, the `Feature` column holds these names, else row indices.
|
|
169
|
+
|
|
170
|
+
Returns (DataFrame, levels). The DataFrame has columns
|
|
171
|
+
Feature, stat, p, (log2FC for welch), FDR_BH,
|
|
172
|
+
sorted by ascending p.
|
|
173
|
+
"""
|
|
174
|
+
X = np.asarray(X, dtype=float)
|
|
175
|
+
groups = list(groups)
|
|
176
|
+
levels = list(dict.fromkeys(groups))
|
|
177
|
+
n = X.shape[0]
|
|
178
|
+
names = list(feature_names) if feature_names is not None else list(range(n))
|
|
179
|
+
res = []
|
|
180
|
+
if method in ("anova", "kruskal"):
|
|
181
|
+
for i in range(n):
|
|
182
|
+
cols = [X[i, np.array(groups) == lv] for lv in levels]
|
|
183
|
+
if any(c.size < 2 for c in cols):
|
|
184
|
+
stat, p = np.nan, np.nan
|
|
185
|
+
elif method == "anova":
|
|
186
|
+
stat, p = stats.f_oneway(*cols)
|
|
187
|
+
else:
|
|
188
|
+
stat, p = stats.kruskal(*cols)
|
|
189
|
+
res.append(dict(Feature=names[i], stat=stat, p=p))
|
|
190
|
+
elif method == "welch":
|
|
191
|
+
if base not in levels or not treats:
|
|
192
|
+
raise ValueError("Welch test needs `base` and non-empty `treats`.")
|
|
193
|
+
b = np.array(groups) == base
|
|
194
|
+
a = np.isin(np.array(groups), treats)
|
|
195
|
+
for i in range(n):
|
|
196
|
+
if X[i, a].size < 2 or X[i, b].size < 2:
|
|
197
|
+
stat, p = np.nan, np.nan
|
|
198
|
+
fc = np.nan
|
|
199
|
+
else:
|
|
200
|
+
stat, p = stats.ttest_ind(X[i, a], X[i, b], equal_var=False)
|
|
201
|
+
fc = X[i, a].mean() - X[i, b].mean()
|
|
202
|
+
res.append(dict(Feature=names[i], stat=stat, p=p, log2FC=fc))
|
|
203
|
+
else:
|
|
204
|
+
raise ValueError(f"unknown method {method!r}")
|
|
205
|
+
|
|
206
|
+
df = pd.DataFrame(res)
|
|
207
|
+
df["p"] = pd.Series(df["p"]).astype(float)
|
|
208
|
+
df = df.sort_values("p", na_position="last").reset_index(drop=True)
|
|
209
|
+
df["FDR_BH"] = bh_fdr(df["p"].values)
|
|
210
|
+
return df, levels
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def add_group_means(df, X, groups, levels, feature_names=None):
|
|
214
|
+
"""Append per-group mean columns (on the supplied, usually scaled, X).
|
|
215
|
+
|
|
216
|
+
`df` must contain a `Feature` column. When `feature_names` is given it
|
|
217
|
+
maps feature names to their row index in X; otherwise `Feature` values
|
|
218
|
+
are assumed to already be row indices.
|
|
219
|
+
"""
|
|
220
|
+
X = np.asarray(X, dtype=float)
|
|
221
|
+
groups = np.asarray(groups)
|
|
222
|
+
out = df.copy()
|
|
223
|
+
if feature_names is not None:
|
|
224
|
+
name_to_row = {n: i for i, n in enumerate(feature_names)}
|
|
225
|
+
else:
|
|
226
|
+
name_to_row = None
|
|
227
|
+
for lv in levels:
|
|
228
|
+
vals = []
|
|
229
|
+
for f in out["Feature"]:
|
|
230
|
+
i = name_to_row.get(f) if name_to_row is not None else f
|
|
231
|
+
vals.append(
|
|
232
|
+
X[i, groups == lv].mean() if (groups == lv).any() else np.nan)
|
|
233
|
+
out[f"mean_{lv}"] = vals
|
|
234
|
+
return out
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def format_p(v):
|
|
238
|
+
"""Human-friendly p-value string (scientific below 1e-4)."""
|
|
239
|
+
if v is None or (isinstance(v, float) and np.isnan(v)):
|
|
240
|
+
return "NA"
|
|
241
|
+
if v < 1e-4:
|
|
242
|
+
return f"{v:.2e}"
|
|
243
|
+
return f"{v:.4f}"
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
# --------------------------------------------------------------------------
|
|
247
|
+
# Volcano (contrast derived from per-group means)
|
|
248
|
+
# --------------------------------------------------------------------------
|
|
249
|
+
|
|
250
|
+
def volcano_table(df, group_a, group_b, fdr_col="FDR_BH", alpha=0.05, fc=1.0):
|
|
251
|
+
"""Build a volcano table for the contrast `group_a` vs `group_b`.
|
|
252
|
+
|
|
253
|
+
Operates on a univariate results frame (as produced by `univariate` +
|
|
254
|
+
`add_group_means`): it needs a `Feature` column, the two `mean_<group>`
|
|
255
|
+
columns and an FDR column. Group means live in the (log2) space the data
|
|
256
|
+
was supplied in, so the fold change is their difference:
|
|
257
|
+
|
|
258
|
+
log2FC = mean_a - mean_b (positive => higher in group_a)
|
|
259
|
+
|
|
260
|
+
Returns a DataFrame with columns
|
|
261
|
+
Feature, log2FC, FDR_BH, neglog10FDR, direction,
|
|
262
|
+
where direction is "up" (FDR < alpha and log2FC >= fc),
|
|
263
|
+
"down"(FDR < alpha and log2FC <= -fc),
|
|
264
|
+
"ns" otherwise.
|
|
265
|
+
"""
|
|
266
|
+
ca, cb = f"mean_{group_a}", f"mean_{group_b}"
|
|
267
|
+
for c in ("Feature", ca, cb, fdr_col):
|
|
268
|
+
if c not in df.columns:
|
|
269
|
+
raise ValueError(f"volcano needs column {c!r}")
|
|
270
|
+
lfc = df[ca].to_numpy(float) - df[cb].to_numpy(float)
|
|
271
|
+
fdr = df[fdr_col].to_numpy(float)
|
|
272
|
+
out = pd.DataFrame({
|
|
273
|
+
"Feature": df["Feature"].to_numpy(),
|
|
274
|
+
"log2FC": lfc,
|
|
275
|
+
"FDR_BH": fdr,
|
|
276
|
+
"neglog10FDR": -np.log10(np.clip(fdr, np.finfo(float).tiny, None)),
|
|
277
|
+
})
|
|
278
|
+
sig = fdr < alpha
|
|
279
|
+
out["direction"] = np.where(
|
|
280
|
+
sig & (lfc >= fc), "up",
|
|
281
|
+
np.where(sig & (lfc <= -fc), "down", "ns"))
|
|
282
|
+
return out
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Widget definitions for wellerlab.metabo."""
|
|
2
|
+
|
|
3
|
+
from .owfeatureimport import OWFeatureImport # noqa: F401
|
|
4
|
+
from .owpreprocess import OWMetaboPreprocess # noqa: F401
|
|
5
|
+
from .owfeaturefilter import OWFeatureFilter # noqa: F401
|
|
6
|
+
from .owunivariate import OWUnivariateStats # noqa: F401
|
|
7
|
+
from .owheatmap import OWMetaboHeatmap # noqa: F401
|
|
8
|
+
from .owvolcano import OWVolcano # noqa: F401
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"OWFeatureImport",
|
|
12
|
+
"OWMetaboPreprocess",
|
|
13
|
+
"OWFeatureFilter",
|
|
14
|
+
"OWUnivariateStats",
|
|
15
|
+
"OWMetaboHeatmap",
|
|
16
|
+
"OWVolcano",
|
|
17
|
+
]
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Small Qt dialog helpers shared by the Metabo widgets.
|
|
2
|
+
|
|
3
|
+
Orange's ``Orange.widgets.utils.filedialogs`` API is not stable across
|
|
4
|
+
releases: the ``OpenFileDialog`` / ``SaveFileDialog`` classes used by older
|
|
5
|
+
widgets no longer exist in current Orange (they were replaced by the
|
|
6
|
+
``open_filename_dialog`` / ``open_filename_dialog_save`` functions). To stay
|
|
7
|
+
working on every Orange version the widgets call plain ``QFileDialog``
|
|
8
|
+
directly through these two helpers — one idiom, no version-specific imports.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def open_feature_table(parent, start_dir=""):
|
|
13
|
+
"""Ask for a feature-table CSV/TXT. Returns a path or None (cancelled)."""
|
|
14
|
+
from AnyQt.QtWidgets import QFileDialog
|
|
15
|
+
path, _ = QFileDialog.getOpenFileName(
|
|
16
|
+
parent, "Open feature table", start_dir or "",
|
|
17
|
+
"Feature table (*.csv *.txt *.tsv);;All files (*)")
|
|
18
|
+
return path or None
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def save_figure(parent, fig, kind="PNG"):
|
|
22
|
+
"""Ask for a path and write figure `fig` as PNG (dpi 300) or SVG.
|
|
23
|
+
|
|
24
|
+
`kind` is 'PNG' or 'SVG' (case-insensitive). Returns the path written,
|
|
25
|
+
or None if the user cancelled.
|
|
26
|
+
"""
|
|
27
|
+
from AnyQt.QtWidgets import QFileDialog
|
|
28
|
+
kind = kind.upper()
|
|
29
|
+
ext = ".png" if kind == "PNG" else ".svg"
|
|
30
|
+
path, _ = QFileDialog.getSaveFileName(
|
|
31
|
+
parent, f"Save figure ({kind})", "", f"{kind} (*{ext});;All files (*)")
|
|
32
|
+
if not path:
|
|
33
|
+
return None
|
|
34
|
+
if not path.lower().endswith(ext):
|
|
35
|
+
path += ext
|
|
36
|
+
fig.savefig(path, dpi=300 if kind == "PNG" else None, facecolor="white")
|
|
37
|
+
return path
|