wellerlab 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. wellerlab-0.1.0/PKG-INFO +77 -0
  2. wellerlab-0.1.0/README.md +62 -0
  3. wellerlab-0.1.0/setup.cfg +4 -0
  4. wellerlab-0.1.0/setup.py +54 -0
  5. wellerlab-0.1.0/wellerlab/__init__.py +17 -0
  6. wellerlab-0.1.0/wellerlab/metabo/__init__.py +3 -0
  7. wellerlab-0.1.0/wellerlab/metabo/metabo_core.py +282 -0
  8. wellerlab-0.1.0/wellerlab/metabo/widgets/__init__.py +17 -0
  9. wellerlab-0.1.0/wellerlab/metabo/widgets/_dialogs.py +37 -0
  10. wellerlab-0.1.0/wellerlab/metabo/widgets/_plot.py +194 -0
  11. wellerlab-0.1.0/wellerlab/metabo/widgets/icons/FeatureFilter.svg +6 -0
  12. wellerlab-0.1.0/wellerlab/metabo/widgets/icons/FeatureImport.svg +6 -0
  13. wellerlab-0.1.0/wellerlab/metabo/widgets/icons/Heatmap.svg +6 -0
  14. wellerlab-0.1.0/wellerlab/metabo/widgets/icons/Preprocess.svg +6 -0
  15. wellerlab-0.1.0/wellerlab/metabo/widgets/icons/UnivariateStats.svg +6 -0
  16. wellerlab-0.1.0/wellerlab/metabo/widgets/icons/Volcano.svg +6 -0
  17. wellerlab-0.1.0/wellerlab/metabo/widgets/owfeaturefilter.py +137 -0
  18. wellerlab-0.1.0/wellerlab/metabo/widgets/owfeatureimport.py +118 -0
  19. wellerlab-0.1.0/wellerlab/metabo/widgets/owheatmap.py +267 -0
  20. wellerlab-0.1.0/wellerlab/metabo/widgets/owpreprocess.py +164 -0
  21. wellerlab-0.1.0/wellerlab/metabo/widgets/owunivariate.py +185 -0
  22. wellerlab-0.1.0/wellerlab/metabo/widgets/owvolcano.py +512 -0
  23. wellerlab-0.1.0/wellerlab/nmr/__init__.py +14 -0
  24. wellerlab-0.1.0/wellerlab/nmr/widgets/__init__.py +17 -0
  25. wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRBaseline.svg +6 -0
  26. wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRBinning.svg +6 -0
  27. wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRExclude.svg +6 -0
  28. wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRFilter.svg +6 -0
  29. wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRNormalize.svg +6 -0
  30. wellerlab-0.1.0/wellerlab/nmr/widgets/icons/NMRReference.svg +6 -0
  31. wellerlab-0.1.0/wellerlab/nmr/widgets/nmr_utils.py +74 -0
  32. wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrbaseline.py +149 -0
  33. wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrbinning.py +124 -0
  34. wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrexclude.py +145 -0
  35. wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrfilter.py +126 -0
  36. wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrnormalize.py +134 -0
  37. wellerlab-0.1.0/wellerlab/nmr/widgets/ownmrreference.py +145 -0
  38. wellerlab-0.1.0/wellerlab/pca/__init__.py +16 -0
  39. wellerlab-0.1.0/wellerlab/pca/pca_analysis.py +246 -0
  40. wellerlab-0.1.0/wellerlab/pca/widgets/__init__.py +7 -0
  41. wellerlab-0.1.0/wellerlab/pca/widgets/icons/PCAPro.svg +6 -0
  42. wellerlab-0.1.0/wellerlab/pca/widgets/owpcawell.py +563 -0
  43. wellerlab-0.1.0/wellerlab/plsda/__init__.py +29 -0
  44. wellerlab-0.1.0/wellerlab/plsda/opls_core.py +377 -0
  45. wellerlab-0.1.0/wellerlab/plsda/oplsda_learner.py +209 -0
  46. wellerlab-0.1.0/wellerlab/plsda/plsda_learner.py +120 -0
  47. wellerlab-0.1.0/wellerlab/plsda/widgets/__init__.py +8 -0
  48. wellerlab-0.1.0/wellerlab/plsda/widgets/icons/OPLSDA.svg +6 -0
  49. wellerlab-0.1.0/wellerlab/plsda/widgets/icons/PLSDA.svg +6 -0
  50. wellerlab-0.1.0/wellerlab/plsda/widgets/owoplsda.py +506 -0
  51. wellerlab-0.1.0/wellerlab/plsda/widgets/owplsda.py +192 -0
  52. wellerlab-0.1.0/wellerlab/widgets/__init__.py +17 -0
  53. wellerlab-0.1.0/wellerlab.egg-info/PKG-INFO +77 -0
  54. wellerlab-0.1.0/wellerlab.egg-info/SOURCES.txt +56 -0
  55. wellerlab-0.1.0/wellerlab.egg-info/dependency_links.txt +1 -0
  56. wellerlab-0.1.0/wellerlab.egg-info/entry_points.txt +2 -0
  57. wellerlab-0.1.0/wellerlab.egg-info/requires.txt +6 -0
  58. wellerlab-0.1.0/wellerlab.egg-info/top_level.txt +1 -0
@@ -0,0 +1,77 @@
1
+ Metadata-Version: 2.1
2
+ Name: wellerlab
3
+ Version: 0.1.0
4
+ Summary: WellerLab Orange3 suite: Metabo statistics, PLS-DA/OPLS-DA, PCA Pro and NMR preprocessing in one category
5
+ Author: Philipp Weller
6
+ Author-email: philipp.weller@googlemail.com
7
+ Project-URL: Source, https://github.com/philippweller/WellerLab/tree/main/orange-wellerlab-addon
8
+ Project-URL: Bug Tracker, https://github.com/philippweller/WellerLab/issues
9
+ Classifier: Development Status :: 3 - Alpha
10
+ Classifier: Intended Audience :: Science/Research
11
+ Classifier: License :: OSI Approved :: MIT License
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Topic :: Scientific/Engineering
14
+ Description-Content-Type: text/markdown
15
+
16
+ # wellerlab — WellerLab Orange3 Suitepack
17
+
18
+ Alle Werkzeuge des Weller-Labors in **einem** Orange3-Add-on und **einer**
19
+ Kategorie: **Weller Lab**.
20
+
21
+ | Familie | Widgets | Inhalt |
22
+ |---|---|---|
23
+ | `wellerlab.metabo` | Metabo Feature Table, Preprocess, Feature Filter, Univariate Stats, Heatmap, Volcano | MetaboAnalyst-artige Statistik für GC-MS/GC-IMS-Feature-Tables (Compound Discoverer). Qt-freier Kern `metabo_core`. |
24
+ | `wellerlab.plsda` | PLS-DA, OPLS-DA | Multivariate Klassifikation; OPLS-DA mit S-Plot (p1/p(corr)), Schwellen, Farbcodierung, Klick/Shift-Klick-Selektion, VIP + orthoVIP, R2X/R2Y/Q2 (7-fach-CV) und Permutationstest. Qt-freier Kern `opls_core`. |
25
+ | `wellerlab.pca` | PCA Pro | Chemometrische PCA: Scaling (none/center/Pareto/autoscale), Komponentenwahl (Kaiser, Varianzanteil, feste Zahl), Hotelling-T²- und Q-Residuen-Diagnostik. |
26
+ | `wellerlab.nmr` | NMR Baseline Correction, Binning (Bucketing), Region Exclusion, Filter (Savitzky-Golay), Normalization, Reference & Alignment | Vorverarbeitung von NMR-Spektren. |
27
+
28
+ ## Installation
29
+
30
+ ```bash
31
+ # aus dem Monorepo (empfohlen): findet Oranges eigenes Python
32
+ python orange-install.py wellerlab
33
+
34
+ # oder direkt
35
+ /Applications/Orange.app/Contents/MacOS/python -m pip install wellerlab
36
+ ```
37
+
38
+ Beim ersten Start danach erscheint die Kategorie **Weller Lab** mit allen
39
+ Widgets. Wichtig: frühere Einzelpakete (`orangemetabo`, `orangeplsda`,
40
+ `orangepca`, `orangenmr`) müssen entfernt sein, sonst erscheinen die Widgets
41
+ doppelt.
42
+
43
+ ## Icons
44
+
45
+ Ein generiertes, konsistentes Set: gleiche Kachelform und Strichstärke,
46
+ Familienfarbe und -glyphe (Chromatogramm = Metabo, Multiplett = NMR,
47
+ latente Ellipsen = PLS/OPLS, Score-Plot = PCA) plus Kurzlabel.
48
+ Erzeugt von `tools/gen_icons.py` im Monorepo.
49
+
50
+ ## Tests (headless, offscreen)
51
+
52
+ Kanonischer Befehl — findet Oranges Python selbst und läuft mit *jedem* Python:
53
+
54
+ ```bash
55
+ python3 run_tests.py # alle Suiten
56
+ python3 run_tests.py --python /pfad/zum/python
57
+ ```
58
+
59
+ Die drei Suiten einzeln:
60
+
61
+ ```bash
62
+ PY=/Applications/Orange.app/Contents/MacOS/python
63
+ $PY _test_opls_core.py # Numerik: R2Y/Q2/VIP/Permutation
64
+ $PY _test_owoplsda.py # Widget: Schwellen, Farben, Selektion, Ausgänge
65
+ $PY _test_suite.py # Integration: alle 15 Widgets, Icons, Kategorie
66
+ ```
67
+
68
+ > **macOS/Apple-Silicon-Falle:** Wird Oranges universeller Interpreter von einem
69
+ > x86_64-Python (Rosetta, z. B. einer Intel-conda-Installation) gestartet, läuft er
70
+ > selbst als x86_64 — dort scheitert der Import von Oranges arm64-only
71
+ > numpy-Extension („you should not try to import numpy from its source
72
+ > directory“). `run_tests.py` umgeht das automatisch mit `arch -arm64`; bei
73
+ > direkten Aufrufen nativ starten oder ebenfalls `arch -arm64` voranstellen.
74
+
75
+ ## Lizenz
76
+
77
+ MIT · Philipp Weller · philipp.weller@googlemail.com
@@ -0,0 +1,62 @@
1
+ # wellerlab — WellerLab Orange3 Suitepack
2
+
3
+ Alle Werkzeuge des Weller-Labors in **einem** Orange3-Add-on und **einer**
4
+ Kategorie: **Weller Lab**.
5
+
6
+ | Familie | Widgets | Inhalt |
7
+ |---|---|---|
8
+ | `wellerlab.metabo` | Metabo Feature Table, Preprocess, Feature Filter, Univariate Stats, Heatmap, Volcano | MetaboAnalyst-artige Statistik für GC-MS/GC-IMS-Feature-Tables (Compound Discoverer). Qt-freier Kern `metabo_core`. |
9
+ | `wellerlab.plsda` | PLS-DA, OPLS-DA | Multivariate Klassifikation; OPLS-DA mit S-Plot (p1/p(corr)), Schwellen, Farbcodierung, Klick/Shift-Klick-Selektion, VIP + orthoVIP, R2X/R2Y/Q2 (7-fach-CV) und Permutationstest. Qt-freier Kern `opls_core`. |
10
+ | `wellerlab.pca` | PCA Pro | Chemometrische PCA: Scaling (none/center/Pareto/autoscale), Komponentenwahl (Kaiser, Varianzanteil, feste Zahl), Hotelling-T²- und Q-Residuen-Diagnostik. |
11
+ | `wellerlab.nmr` | NMR Baseline Correction, Binning (Bucketing), Region Exclusion, Filter (Savitzky-Golay), Normalization, Reference & Alignment | Vorverarbeitung von NMR-Spektren. |
12
+
13
+ ## Installation
14
+
15
+ ```bash
16
+ # aus dem Monorepo (empfohlen): findet Oranges eigenes Python
17
+ python orange-install.py wellerlab
18
+
19
+ # oder direkt
20
+ /Applications/Orange.app/Contents/MacOS/python -m pip install wellerlab
21
+ ```
22
+
23
+ Beim ersten Start danach erscheint die Kategorie **Weller Lab** mit allen
24
+ Widgets. Wichtig: frühere Einzelpakete (`orangemetabo`, `orangeplsda`,
25
+ `orangepca`, `orangenmr`) müssen entfernt sein, sonst erscheinen die Widgets
26
+ doppelt.
27
+
28
+ ## Icons
29
+
30
+ Ein generiertes, konsistentes Set: gleiche Kachelform und Strichstärke,
31
+ Familienfarbe und -glyphe (Chromatogramm = Metabo, Multiplett = NMR,
32
+ latente Ellipsen = PLS/OPLS, Score-Plot = PCA) plus Kurzlabel.
33
+ Erzeugt von `tools/gen_icons.py` im Monorepo.
34
+
35
+ ## Tests (headless, offscreen)
36
+
37
+ Kanonischer Befehl — findet Oranges Python selbst und läuft mit *jedem* Python:
38
+
39
+ ```bash
40
+ python3 run_tests.py # alle Suiten
41
+ python3 run_tests.py --python /pfad/zum/python
42
+ ```
43
+
44
+ Die drei Suiten einzeln:
45
+
46
+ ```bash
47
+ PY=/Applications/Orange.app/Contents/MacOS/python
48
+ $PY _test_opls_core.py # Numerik: R2Y/Q2/VIP/Permutation
49
+ $PY _test_owoplsda.py # Widget: Schwellen, Farben, Selektion, Ausgänge
50
+ $PY _test_suite.py # Integration: alle 15 Widgets, Icons, Kategorie
51
+ ```
52
+
53
+ > **macOS/Apple-Silicon-Falle:** Wird Oranges universeller Interpreter von einem
54
+ > x86_64-Python (Rosetta, z. B. einer Intel-conda-Installation) gestartet, läuft er
55
+ > selbst als x86_64 — dort scheitert der Import von Oranges arm64-only
56
+ > numpy-Extension („you should not try to import numpy from its source
57
+ > directory“). `run_tests.py` umgeht das automatisch mit `arch -arm64`; bei
58
+ > direkten Aufrufen nativ starten oder ebenfalls `arch -arm64` voranstellen.
59
+
60
+ ## Lizenz
61
+
62
+ MIT · Philipp Weller · philipp.weller@googlemail.com
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,54 @@
1
+ from pathlib import Path
2
+
3
+ from setuptools import setup, find_packages
4
+
5
+ HERE = Path(__file__).parent
6
+
7
+ README = HERE / "README.md"
8
+ long_description = (README.read_text(encoding="utf-8") if README.exists()
9
+ else "WellerLab Orange3 tool suite: metabolomics feature-table "
10
+ "statistics, PLS-DA/OPLS-DA, PCA Pro and NMR preprocessing.")
11
+
12
+ setup(
13
+ name="wellerlab",
14
+ version="0.1.0",
15
+ description="WellerLab Orange3 suite: Metabo statistics, PLS-DA/OPLS-DA, "
16
+ "PCA Pro and NMR preprocessing in one category",
17
+ long_description=long_description,
18
+ long_description_content_type="text/markdown",
19
+ packages=find_packages(include=["wellerlab", "wellerlab.*"]),
20
+ include_package_data=True,
21
+ author="Philipp Weller",
22
+ author_email="philipp.weller@googlemail.com",
23
+ install_requires=[
24
+ "Orange3>=3.40.0",
25
+ "numpy",
26
+ "scipy",
27
+ "matplotlib",
28
+ "scikit-learn",
29
+ "pandas",
30
+ ],
31
+ entry_points={
32
+ "orange.widgets": (
33
+ "Weller Lab = wellerlab.widgets",
34
+ ),
35
+ },
36
+ package_data={
37
+ "wellerlab": ["widgets/icons/*.svg",
38
+ "metabo/widgets/icons/*.svg",
39
+ "plsda/widgets/icons/*.svg",
40
+ "pca/widgets/icons/*.svg",
41
+ "nmr/widgets/icons/*.svg"],
42
+ },
43
+ project_urls={
44
+ "Source": "https://github.com/philippweller/WellerLab/tree/main/orange-wellerlab-addon",
45
+ "Bug Tracker": "https://github.com/philippweller/WellerLab/issues",
46
+ },
47
+ classifiers=[
48
+ "Development Status :: 3 - Alpha",
49
+ "Intended Audience :: Science/Research",
50
+ "License :: OSI Approved :: MIT License",
51
+ "Programming Language :: Python :: 3",
52
+ "Topic :: Scientific/Engineering",
53
+ ],
54
+ )
@@ -0,0 +1,17 @@
1
+ """
2
+ WellerLab — the complete Orange3 tool suite of the Weller lab.
3
+
4
+ One distribution, one Orange category ("Weller Lab"):
5
+
6
+ wellerlab.metabo MetaboAnalyst-style feature-table statistics
7
+ (import, preprocess, filter, univariate, heatmap, volcano)
8
+ wellerlab.plsda PLS-DA and OPLS-DA (S-plot, VIP, Q2, permutation)
9
+ wellerlab.pca PCA Pro
10
+ wellerlab.nmr NMR preprocessing (baseline, binning, exclusion,
11
+ filter, normalization, reference)
12
+
13
+ The numerical cores (``metabo.metabo_core``, ``plsda.opls_core``) are free of
14
+ Qt/Orange and can be imported and tested headlessly.
15
+ """
16
+
17
+ __version__ = "0.1.0"
@@ -0,0 +1,3 @@
1
+ """wellerlab.metabo — MetaboAnalyst-style add-on for Orange3."""
2
+
3
+ __version__ = "0.4.4"
@@ -0,0 +1,282 @@
1
+ #!/usr/bin/env python
2
+ # -*- coding: utf-8 -*-
3
+ """
4
+ MetaboAnalyst-style univariate statistics core for GC-MS / GC-IMS
5
+ feature tables (Compound Discoverer exports).
6
+
7
+ This module is the analytics heart of the wellerlab.metabo add-on. It is
8
+ deliberately free of any Qt / Orange dependency so it can be unit-tested
9
+ headlessly with plain numpy/scipy. The widgets in `orangemetabo/widgets/`
10
+ are thin GUI wrappers around the functions here.
11
+
12
+ Pipeline (matches the validated ground truth `cv_anova_alle_97_features.csv`,
13
+ 97/97 features on F, p and FDR_BH):
14
+
15
+ load -> sum-normalise -> log2 -> autoscale (z-score per feature)
16
+ -> univariate statistics (one-way ANOVA / Welch t / Kruskal-Wallis)
17
+ -> Benjamini-Hochberg FDR
18
+
19
+ All functions are pure: they take arrays / lists and return DataFrames or
20
+ arrays. No file I/O except the optional `load_feature_table` reader.
21
+ """
22
+ import csv
23
+ import numpy as np
24
+ import pandas as pd
25
+ from scipy import stats
26
+
27
+
28
+ # --------------------------------------------------------------------------
29
+ # Feature-table reader (Compound Discoverer 2-header semicolon CSV)
30
+ # --------------------------------------------------------------------------
31
+
32
+ def load_feature_table(path):
33
+ """Read a Compound Discoverer feature-table CSV.
34
+
35
+ Expected layout (semicolon-separated, UTF-8-BOM):
36
+ row 0: "" , "Area: <sample>.raw", ...
37
+ row 1: "" , "<group>", ...
38
+ row 2+: "<feature name>", values...
39
+
40
+ Returns (feature_names: list[str], samples: list[str],
41
+ groups: list[str], X: np.ndarray[features x samples]).
42
+ """
43
+ with open(path, encoding="utf-8-sig", newline="") as f:
44
+ rows = list(csv.reader(f, delimiter=";"))
45
+ samples = [h.replace("Area: ", "").replace(".raw", "").strip()
46
+ for h in rows[0][1:]]
47
+ groups = [g.strip() for g in rows[1][1:]]
48
+ feat = [r[0].strip() for r in rows[2:]]
49
+ X = np.array(
50
+ [[float(v) for v in r[1:1 + len(samples)]] for r in rows[2:]],
51
+ dtype=float,
52
+ )
53
+ return feat, samples, groups, X
54
+
55
+
56
+ # --------------------------------------------------------------------------
57
+ # Preprocessing
58
+ # --------------------------------------------------------------------------
59
+
60
+ def normalize_sum(X):
61
+ """Total-area (sum) normalisation: rescale each sample column so that
62
+ every column has the mean of the column totals."""
63
+ col_sums = X.sum(axis=0)
64
+ return X / col_sums * col_sums.mean()
65
+
66
+
67
+ def log2_transform(X):
68
+ """log2 with a guard against non-positive values (clip to tiny eps)."""
69
+ eps = np.finfo(float).tiny
70
+ return np.log2(np.clip(X, eps, None))
71
+
72
+
73
+ def impute_low(X, threshold=0.05, method="knn", k=3):
74
+ """Impute below-threshold values (fraction of column min or a hard cap).
75
+
76
+ Methods:
77
+ 'constant' : replace with the column minimum (or 0).
78
+ 'min' : replace with the column minimum.
79
+ 'knn' : replace with the mean of the k nearest samples in the
80
+ Euclidean distance of the non-imputed rows.
81
+ Returns (X_imputed, imputed_mask).
82
+ """
83
+ X = np.asarray(X, dtype=float).copy()
84
+ col_min = X.min(axis=0)
85
+ mask = X < (col_min * (1 + threshold))
86
+ if not mask.any():
87
+ return X, mask
88
+ if method in ("constant", "min"):
89
+ rows, cols = np.nonzero(mask)
90
+ X[rows, cols] = col_min[cols]
91
+ return X, mask
92
+ # knn: distance between samples (columns) over all feature rows;
93
+ # impute each flagged value with the mean of its k nearest samples.
94
+ n_rows, n_cols = X.shape
95
+ d = np.sqrt(((X[:, :, None] - X[:, None, :]) ** 2).sum(axis=0)) # c x c
96
+ np.fill_diagonal(d, np.inf)
97
+ order = d.argsort(axis=1)
98
+ for c in range(n_cols):
99
+ bad = np.nonzero(mask[:, c])[0]
100
+ if bad.size == 0:
101
+ continue
102
+ for r in bad:
103
+ # prefer samples that are not flagged on this same feature
104
+ nb = order[c, :max(k, 1)]
105
+ ok = [j for j in nb if not (mask[r, j] and j != c)]
106
+ if not ok:
107
+ X[r, c] = col_min[c]
108
+ else:
109
+ X[r, c] = X[r, ok].mean()
110
+ return X, mask
111
+
112
+
113
+ def scale_rows(X, method="autoscale"):
114
+ """Per-feature (row) scaling.
115
+
116
+ 'autoscale' : z-score (mean 0, std 1, ddof=1)
117
+ 'pareto' : (x - mean) / sqrt(std)
118
+ 'none' : unchanged
119
+ Constant rows are returned as zeros (division guard).
120
+ """
121
+ X = np.asarray(X, dtype=float)
122
+ mean = X.mean(axis=1, keepdims=True)
123
+ if method == "none":
124
+ return X
125
+ std = X.std(axis=1, keepdims=True, ddof=1)
126
+ denom = np.where(std > 0, np.sqrt(std) if method == "pareto" else std, 1.0)
127
+ return (X - mean) / denom
128
+
129
+
130
+ # --------------------------------------------------------------------------
131
+ # Multiple-testing correction
132
+ # --------------------------------------------------------------------------
133
+
134
+ def bh_fdr(p):
135
+ """Benjamini-Hochberg FDR for an array of p-values."""
136
+ p = np.asarray(p, dtype=float)
137
+ m = len(p)
138
+ if m == 0:
139
+ return p
140
+ order = p.argsort()
141
+ ranked = p[order]
142
+ q = np.minimum.accumulate((ranked * m / np.arange(1, m + 1))[::-1])[::-1]
143
+ qf = np.empty(m)
144
+ qf[order] = np.clip(q, 0, 1)
145
+ return qf
146
+
147
+
148
+ # --------------------------------------------------------------------------
149
+ # Univariate statistics
150
+ # --------------------------------------------------------------------------
151
+
152
+ def _groups_arrays(X, groups, levels):
153
+ """Yield the per-group column slices for every feature row."""
154
+ gi = {lv: np.array(groups) == lv for lv in levels}
155
+ return [X[:, idx] for idx in gi.values()]
156
+
157
+
158
+ def univariate(X, groups, method="anova", base=None, treats=None,
159
+ feature_names=None):
160
+ """Run the selected univariate test for every feature row.
161
+
162
+ method:
163
+ 'anova' : one-way ANOVA across all levels (>= 2 levels, >= 2 per level).
164
+ 'welch' : Welch two-sample t-test, `treats` combined vs `base`.
165
+ 'kruskal': one-way Kruskal-Wallis across all levels.
166
+
167
+ feature_names: optional list of row labels (same length as X rows);
168
+ when given, the `Feature` column holds these names, else row indices.
169
+
170
+ Returns (DataFrame, levels). The DataFrame has columns
171
+ Feature, stat, p, (log2FC for welch), FDR_BH,
172
+ sorted by ascending p.
173
+ """
174
+ X = np.asarray(X, dtype=float)
175
+ groups = list(groups)
176
+ levels = list(dict.fromkeys(groups))
177
+ n = X.shape[0]
178
+ names = list(feature_names) if feature_names is not None else list(range(n))
179
+ res = []
180
+ if method in ("anova", "kruskal"):
181
+ for i in range(n):
182
+ cols = [X[i, np.array(groups) == lv] for lv in levels]
183
+ if any(c.size < 2 for c in cols):
184
+ stat, p = np.nan, np.nan
185
+ elif method == "anova":
186
+ stat, p = stats.f_oneway(*cols)
187
+ else:
188
+ stat, p = stats.kruskal(*cols)
189
+ res.append(dict(Feature=names[i], stat=stat, p=p))
190
+ elif method == "welch":
191
+ if base not in levels or not treats:
192
+ raise ValueError("Welch test needs `base` and non-empty `treats`.")
193
+ b = np.array(groups) == base
194
+ a = np.isin(np.array(groups), treats)
195
+ for i in range(n):
196
+ if X[i, a].size < 2 or X[i, b].size < 2:
197
+ stat, p = np.nan, np.nan
198
+ fc = np.nan
199
+ else:
200
+ stat, p = stats.ttest_ind(X[i, a], X[i, b], equal_var=False)
201
+ fc = X[i, a].mean() - X[i, b].mean()
202
+ res.append(dict(Feature=names[i], stat=stat, p=p, log2FC=fc))
203
+ else:
204
+ raise ValueError(f"unknown method {method!r}")
205
+
206
+ df = pd.DataFrame(res)
207
+ df["p"] = pd.Series(df["p"]).astype(float)
208
+ df = df.sort_values("p", na_position="last").reset_index(drop=True)
209
+ df["FDR_BH"] = bh_fdr(df["p"].values)
210
+ return df, levels
211
+
212
+
213
+ def add_group_means(df, X, groups, levels, feature_names=None):
214
+ """Append per-group mean columns (on the supplied, usually scaled, X).
215
+
216
+ `df` must contain a `Feature` column. When `feature_names` is given it
217
+ maps feature names to their row index in X; otherwise `Feature` values
218
+ are assumed to already be row indices.
219
+ """
220
+ X = np.asarray(X, dtype=float)
221
+ groups = np.asarray(groups)
222
+ out = df.copy()
223
+ if feature_names is not None:
224
+ name_to_row = {n: i for i, n in enumerate(feature_names)}
225
+ else:
226
+ name_to_row = None
227
+ for lv in levels:
228
+ vals = []
229
+ for f in out["Feature"]:
230
+ i = name_to_row.get(f) if name_to_row is not None else f
231
+ vals.append(
232
+ X[i, groups == lv].mean() if (groups == lv).any() else np.nan)
233
+ out[f"mean_{lv}"] = vals
234
+ return out
235
+
236
+
237
+ def format_p(v):
238
+ """Human-friendly p-value string (scientific below 1e-4)."""
239
+ if v is None or (isinstance(v, float) and np.isnan(v)):
240
+ return "NA"
241
+ if v < 1e-4:
242
+ return f"{v:.2e}"
243
+ return f"{v:.4f}"
244
+
245
+
246
+ # --------------------------------------------------------------------------
247
+ # Volcano (contrast derived from per-group means)
248
+ # --------------------------------------------------------------------------
249
+
250
+ def volcano_table(df, group_a, group_b, fdr_col="FDR_BH", alpha=0.05, fc=1.0):
251
+ """Build a volcano table for the contrast `group_a` vs `group_b`.
252
+
253
+ Operates on a univariate results frame (as produced by `univariate` +
254
+ `add_group_means`): it needs a `Feature` column, the two `mean_<group>`
255
+ columns and an FDR column. Group means live in the (log2) space the data
256
+ was supplied in, so the fold change is their difference:
257
+
258
+ log2FC = mean_a - mean_b (positive => higher in group_a)
259
+
260
+ Returns a DataFrame with columns
261
+ Feature, log2FC, FDR_BH, neglog10FDR, direction,
262
+ where direction is "up" (FDR < alpha and log2FC >= fc),
263
+ "down"(FDR < alpha and log2FC <= -fc),
264
+ "ns" otherwise.
265
+ """
266
+ ca, cb = f"mean_{group_a}", f"mean_{group_b}"
267
+ for c in ("Feature", ca, cb, fdr_col):
268
+ if c not in df.columns:
269
+ raise ValueError(f"volcano needs column {c!r}")
270
+ lfc = df[ca].to_numpy(float) - df[cb].to_numpy(float)
271
+ fdr = df[fdr_col].to_numpy(float)
272
+ out = pd.DataFrame({
273
+ "Feature": df["Feature"].to_numpy(),
274
+ "log2FC": lfc,
275
+ "FDR_BH": fdr,
276
+ "neglog10FDR": -np.log10(np.clip(fdr, np.finfo(float).tiny, None)),
277
+ })
278
+ sig = fdr < alpha
279
+ out["direction"] = np.where(
280
+ sig & (lfc >= fc), "up",
281
+ np.where(sig & (lfc <= -fc), "down", "ns"))
282
+ return out
@@ -0,0 +1,17 @@
1
+ """Widget definitions for wellerlab.metabo."""
2
+
3
+ from .owfeatureimport import OWFeatureImport # noqa: F401
4
+ from .owpreprocess import OWMetaboPreprocess # noqa: F401
5
+ from .owfeaturefilter import OWFeatureFilter # noqa: F401
6
+ from .owunivariate import OWUnivariateStats # noqa: F401
7
+ from .owheatmap import OWMetaboHeatmap # noqa: F401
8
+ from .owvolcano import OWVolcano # noqa: F401
9
+
10
+ __all__ = [
11
+ "OWFeatureImport",
12
+ "OWMetaboPreprocess",
13
+ "OWFeatureFilter",
14
+ "OWUnivariateStats",
15
+ "OWMetaboHeatmap",
16
+ "OWVolcano",
17
+ ]
@@ -0,0 +1,37 @@
1
+ """Small Qt dialog helpers shared by the Metabo widgets.
2
+
3
+ Orange's ``Orange.widgets.utils.filedialogs`` API is not stable across
4
+ releases: the ``OpenFileDialog`` / ``SaveFileDialog`` classes used by older
5
+ widgets no longer exist in current Orange (they were replaced by the
6
+ ``open_filename_dialog`` / ``open_filename_dialog_save`` functions). To stay
7
+ working on every Orange version the widgets call plain ``QFileDialog``
8
+ directly through these two helpers — one idiom, no version-specific imports.
9
+ """
10
+
11
+
12
+ def open_feature_table(parent, start_dir=""):
13
+ """Ask for a feature-table CSV/TXT. Returns a path or None (cancelled)."""
14
+ from AnyQt.QtWidgets import QFileDialog
15
+ path, _ = QFileDialog.getOpenFileName(
16
+ parent, "Open feature table", start_dir or "",
17
+ "Feature table (*.csv *.txt *.tsv);;All files (*)")
18
+ return path or None
19
+
20
+
21
+ def save_figure(parent, fig, kind="PNG"):
22
+ """Ask for a path and write figure `fig` as PNG (dpi 300) or SVG.
23
+
24
+ `kind` is 'PNG' or 'SVG' (case-insensitive). Returns the path written,
25
+ or None if the user cancelled.
26
+ """
27
+ from AnyQt.QtWidgets import QFileDialog
28
+ kind = kind.upper()
29
+ ext = ".png" if kind == "PNG" else ".svg"
30
+ path, _ = QFileDialog.getSaveFileName(
31
+ parent, f"Save figure ({kind})", "", f"{kind} (*{ext});;All files (*)")
32
+ if not path:
33
+ return None
34
+ if not path.lower().endswith(ext):
35
+ path += ext
36
+ fig.savefig(path, dpi=300 if kind == "PNG" else None, facecolor="white")
37
+ return path