plot-misc 2.2.1__py3-none-any.whl → 2.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- plot_misc/_version.py +1 -1
- plot_misc/constants.py +6 -0
- plot_misc/example_data/examples.py +48 -0
- plot_misc/heatmap.py +221 -8
- plot_misc/utils/utils.py +206 -95
- plot_misc/volcano.py +5 -5
- {plot_misc-2.2.1.dist-info → plot_misc-2.2.2.dist-info}/METADATA +22 -8
- {plot_misc-2.2.1.dist-info → plot_misc-2.2.2.dist-info}/RECORD +11 -11
- {plot_misc-2.2.1.dist-info → plot_misc-2.2.2.dist-info}/WHEEL +0 -0
- {plot_misc-2.2.1.dist-info → plot_misc-2.2.2.dist-info}/licenses/LICENSE +0 -0
- {plot_misc-2.2.1.dist-info → plot_misc-2.2.2.dist-info}/top_level.txt +0 -0
plot_misc/_version.py
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
__version__ = '2.2.
|
|
1
|
+
__version__ = '2.2.2'
|
plot_misc/constants.py
CHANGED
|
@@ -58,6 +58,8 @@ class UtilsNames(object):
|
|
|
58
58
|
annot_pval = 'matrix_pvalue'
|
|
59
59
|
annot_effect = 'matrix_point_estimate'
|
|
60
60
|
value_point = 'curated_matrix_point_estimate_value'
|
|
61
|
+
value_unsigned_log = 'curated_matrix_value_unsigned_log'
|
|
62
|
+
value_raw = 'curated_matrix_value_raw'
|
|
61
63
|
value_original = 'crude_point_estimate'
|
|
62
64
|
source_data = 'source_data'
|
|
63
65
|
mat_point = 'point'
|
|
@@ -67,8 +69,12 @@ class UtilsNames(object):
|
|
|
67
69
|
mat_outcome = 'outcome'
|
|
68
70
|
mat_exposure_list = ['IL2ra', 'IP10', 'SCF', 'TRAIL']
|
|
69
71
|
mat_outcome_list = ['HDL-C', 'LDL-C']
|
|
72
|
+
mat_annot_symbol = 'symbol'
|
|
70
73
|
mat_annot_star = 'star'
|
|
71
74
|
mat_annot_pval = 'pvalues'
|
|
75
|
+
mat_annot_pval_signed = 'pvalues_signed'
|
|
76
|
+
mat_annot_pval_unsigned = 'pvalues_unsigned'
|
|
77
|
+
mat_annot_pval_raw = 'pvalues_raw'
|
|
72
78
|
mat_annot_point = 'point_estimates'
|
|
73
79
|
mat_annot_none = '`NoneType`'
|
|
74
80
|
roc_false_positive = 'false_positive'
|
|
@@ -508,6 +508,54 @@ def heatmap_pvalue_matrix(**kwargs):
|
|
|
508
508
|
data.index.name = UtilsNames.mat_outcome
|
|
509
509
|
return data
|
|
510
510
|
|
|
511
|
+
# ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
512
|
+
@dataset
|
|
513
|
+
def qc_matrix(seed=2026):
|
|
514
|
+
"""
|
|
515
|
+
Creates a dummy quality-control (QC) data set to showcase
|
|
516
|
+
`plot_misc.heatmap.masked_heatmap`.
|
|
517
|
+
|
|
518
|
+
The values are signed, standardised QC deviations for a set of samples
|
|
519
|
+
(rows) across several QC metrics (columns). The accompanying indicator
|
|
520
|
+
flags the cells that failed QC (an absolute deviation above 2), i.e. the
|
|
521
|
+
cells `masked_heatmap` should highlight; the passing cells are left to the
|
|
522
|
+
background layer.
|
|
523
|
+
|
|
524
|
+
Parameters
|
|
525
|
+
----------
|
|
526
|
+
seed : `int`, default 2026
|
|
527
|
+
Seed for the random number generator, ensuring a reproducible matrix.
|
|
528
|
+
|
|
529
|
+
Returns
|
|
530
|
+
-------
|
|
531
|
+
values : `pd.DataFrame`
|
|
532
|
+
Signed standardised QC deviations of shape (12, 6).
|
|
533
|
+
indicator : `pd.DataFrame`
|
|
534
|
+
A binary (0/1) table of the same shape as `values`, equal to 1 where
|
|
535
|
+
the metric failed QC and 0 otherwise.
|
|
536
|
+
"""
|
|
537
|
+
rng = np.random.default_rng(seed)
|
|
538
|
+
samples = ['Sample_{:02d}'.format(i) for i in range(1, 13)]
|
|
539
|
+
metrics = [
|
|
540
|
+
'CallRate', 'Heterozygosity', 'Contamination', 'MeanDepth',
|
|
541
|
+
'DuplicationRate', 'InsertSize',
|
|
542
|
+
]
|
|
543
|
+
values = pd.DataFrame(
|
|
544
|
+
rng.normal(loc=0.0, scale=1.3, size=(len(samples), len(metrics))),
|
|
545
|
+
index=samples, columns=metrics,
|
|
546
|
+
)
|
|
547
|
+
# inject a handful of unambiguous QC failures so the showcase always has
|
|
548
|
+
# highlighted cells regardless of the random draw
|
|
549
|
+
values.iloc[0, 2] = 3.4
|
|
550
|
+
values.iloc[3, 0] = -3.1
|
|
551
|
+
values.iloc[5, 4] = 2.8
|
|
552
|
+
values.iloc[7, 1] = -2.6
|
|
553
|
+
values.iloc[9, 5] = 3.0
|
|
554
|
+
values.iloc[11, 3] = -2.9
|
|
555
|
+
values = values.round(3)
|
|
556
|
+
indicator = (values.abs() > 2).astype(int)
|
|
557
|
+
return values, indicator
|
|
558
|
+
|
|
511
559
|
# ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
512
560
|
@dataset
|
|
513
561
|
def load_calibration_data(**kwargs):
|
plot_misc/heatmap.py
CHANGED
|
@@ -11,6 +11,10 @@ heatmap(data, row_labels, col_labels, ...)
|
|
|
11
11
|
Draws a standard heatmap using matplotlib's `imshow`, with options for
|
|
12
12
|
gridlines, tick formatting, and embedded colourbars.
|
|
13
13
|
|
|
14
|
+
masked_heatmap(data, indicator, row_labels, col_labels, ...)
|
|
15
|
+
Draws a two-layer heatmap: a single-colour background and, on top of it,
|
|
16
|
+
the heatmap restricted to the cells flagged by a binary indicator table.
|
|
17
|
+
|
|
14
18
|
annotate_heatmap(im, data=None, valfmt=None, ...)
|
|
15
19
|
Adds text annotations to an existing heatmap image (AxesImage object),
|
|
16
20
|
with configurable formatting and colour thresholding.
|
|
@@ -32,9 +36,13 @@ import numpy as np
|
|
|
32
36
|
import pandas as pd
|
|
33
37
|
import matplotlib
|
|
34
38
|
import matplotlib.pyplot as plt
|
|
39
|
+
from matplotlib.colors import ListedColormap
|
|
40
|
+
from matplotlib.patches import Rectangle
|
|
35
41
|
from plot_misc.utils.utils import _update_kwargs
|
|
36
42
|
from plot_misc.errors import (
|
|
37
43
|
is_type,
|
|
44
|
+
is_df,
|
|
45
|
+
InputValidationError,
|
|
38
46
|
)
|
|
39
47
|
from plot_misc.constants import Real
|
|
40
48
|
from typing import Any
|
|
@@ -45,6 +53,7 @@ def heatmap(data:pd.DataFrame | np.ndarray, row_labels:list[str] | np.ndarray,
|
|
|
45
53
|
grid_linestyle:str='-', grid_linewidth:float=3,
|
|
46
54
|
cbar_bool:bool=False, cbar_label:str="",
|
|
47
55
|
ax:plt.Axes | None = None,
|
|
56
|
+
figsize:tuple[float,float] | None = None,
|
|
48
57
|
grid_kw:dict[Any,Any] | None = None,
|
|
49
58
|
cbar_kw:dict[Any,Any] | None = None,
|
|
50
59
|
**kwargs:Any,
|
|
@@ -78,6 +87,8 @@ def heatmap(data:pd.DataFrame | np.ndarray, row_labels:list[str] | np.ndarray,
|
|
|
78
87
|
ax : `plt.Axes` or `None`, default None
|
|
79
88
|
A `matplotlib.axes.Axes` instance to which the heatmap is plotted. If
|
|
80
89
|
not provided, use current axes or create a new one.
|
|
90
|
+
figsize : `tuple` [`float`, `float`] or `None`, default `None`
|
|
91
|
+
Figure size in inches (width, height). Ignored if `ax` is provided.
|
|
81
92
|
grid_kw : `dict` [`str`,`any`] or `None`, default None
|
|
82
93
|
A dictionary with arguments to `matplotlib.Axes.grid`.
|
|
83
94
|
cbar_kw : `dict` [`str`, `any`] or `None`, default `None`
|
|
@@ -105,10 +116,20 @@ def heatmap(data:pd.DataFrame | np.ndarray, row_labels:list[str] | np.ndarray,
|
|
|
105
116
|
Matplotlib Gallery.
|
|
106
117
|
https://matplotlib.org/stable/gallery/images_contours_and_fields/image_annotated_heatmap.html
|
|
107
118
|
"""
|
|
108
|
-
|
|
119
|
+
# check in put
|
|
120
|
+
is_type(data, (pd.DataFrame, np.ndarray))
|
|
121
|
+
is_type(row_labels, (list, np.ndarray))
|
|
122
|
+
is_type(col_labels, (list, np.ndarray))
|
|
123
|
+
is_type(grid_col, str)
|
|
124
|
+
is_type(grid_linestyle, str)
|
|
125
|
+
is_type(grid_linewidth, Real)
|
|
126
|
+
is_type(cbar_bool, bool)
|
|
127
|
+
is_type(cbar_label, str)
|
|
109
128
|
# create a axes if needed
|
|
110
|
-
if
|
|
111
|
-
ax = plt.
|
|
129
|
+
if ax is None:
|
|
130
|
+
_, ax = plt.subplots(figsize=figsize)
|
|
131
|
+
else:
|
|
132
|
+
f = ax.figure
|
|
112
133
|
# check input
|
|
113
134
|
if isinstance(data, pd.DataFrame):
|
|
114
135
|
matrix = data.copy().to_numpy()
|
|
@@ -117,10 +138,6 @@ def heatmap(data:pd.DataFrame | np.ndarray, row_labels:list[str] | np.ndarray,
|
|
|
117
138
|
# copy
|
|
118
139
|
row_lab = row_labels
|
|
119
140
|
col_lab = col_labels
|
|
120
|
-
# check additional input
|
|
121
|
-
is_type(row_lab, (list, np.array))
|
|
122
|
-
is_type(col_lab, (list, np.array))
|
|
123
|
-
is_type(cbar_label, str)
|
|
124
141
|
# map None to dict
|
|
125
142
|
grid_kw = grid_kw or {}
|
|
126
143
|
cbar_kw = cbar_kw or {}
|
|
@@ -156,12 +173,201 @@ def heatmap(data:pd.DataFrame | np.ndarray, row_labels:list[str] | np.ndarray,
|
|
|
156
173
|
new_grid_kwargs = _update_kwargs(
|
|
157
174
|
update_dict=grid_kw, which="minor", color=grid_col,
|
|
158
175
|
linestyle=grid_linestyle, linewidth=grid_linewidth,
|
|
159
|
-
)
|
|
176
|
+
clip_on=False,)
|
|
160
177
|
ax.grid(**new_grid_kwargs)
|
|
161
178
|
ax.tick_params(which="minor", bottom=False, left=False)
|
|
162
179
|
# return stuff
|
|
163
180
|
return im, cbar
|
|
164
181
|
|
|
182
|
+
# ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
183
|
+
def masked_heatmap(data:pd.DataFrame | np.ndarray,
|
|
184
|
+
indicator:pd.DataFrame | np.ndarray,
|
|
185
|
+
row_labels:list[str] | np.ndarray,
|
|
186
|
+
col_labels:list[str] | np.ndarray,
|
|
187
|
+
background_col:str='white', background_gridcol:str='white',
|
|
188
|
+
background_linestyle:str='-', background_linewidth:float=0.5,
|
|
189
|
+
background_zorder:Real = 1,
|
|
190
|
+
outline_col:str='black', outline_linestyle:str='-',
|
|
191
|
+
outline_linewidth:float=1.5, outline_zorder:Real = 2,
|
|
192
|
+
frame: bool=False,
|
|
193
|
+
cbar_bool:bool=False, cbar_label:str="",
|
|
194
|
+
ax:plt.Axes | None = None,
|
|
195
|
+
figsize:tuple[float,float] | None = None,
|
|
196
|
+
grid_kw:dict[Any,Any] | None = None,
|
|
197
|
+
cbar_kw:dict[Any,Any] | None = None,
|
|
198
|
+
background_kw:dict[Any,Any] | None = None,
|
|
199
|
+
outline_kw:dict[Any,Any] | None = None,
|
|
200
|
+
**kwargs: Any,
|
|
201
|
+
) -> tuple[matplotlib.image.AxesImage,
|
|
202
|
+
matplotlib.colorbar.Colorbar]:
|
|
203
|
+
"""
|
|
204
|
+
Plot a two-layer heatmap masked by a binary indicator table.
|
|
205
|
+
|
|
206
|
+
The function draws two layers. First a single-colour background covering
|
|
207
|
+
every cell (carrying an optional grid lattice). Second, the heatmap of
|
|
208
|
+
`data`, restricted to the cells where `indicator` equals 1.
|
|
209
|
+
|
|
210
|
+
Parameters
|
|
211
|
+
----------
|
|
212
|
+
data : `pd.DataFrame` or `np.ndarray`
|
|
213
|
+
A 2D array of shape (M, N) containing the values to plot.
|
|
214
|
+
indicator : `pd.DataFrame` or `np.ndarray`
|
|
215
|
+
A binary (0/1, booleans accepted) array of the same shape as `data`.
|
|
216
|
+
Only cells equal to 1 are drawn and outlined.
|
|
217
|
+
row_labels : `list` [`str`] or `np.ndarray`
|
|
218
|
+
A list or array of length M with the labels for the rows.
|
|
219
|
+
col_labels : `list` [`str`] or `np.ndarray`
|
|
220
|
+
A list or array of length N with the labels for the columns.
|
|
221
|
+
background_col : `str`, default 'white'
|
|
222
|
+
The fill colour of the background layer.
|
|
223
|
+
background_gridcol : `str`, default 'white'
|
|
224
|
+
The colour of the background grid lattice lines.
|
|
225
|
+
background_linestyle : `str`, default '-'
|
|
226
|
+
The linestyle of the background grid lattice.
|
|
227
|
+
background_linewidth : `float`, default 0.5
|
|
228
|
+
The width of the background grid lattice. Set to 0 to suppress it.
|
|
229
|
+
background_zorder : `int`, `float` default `1`
|
|
230
|
+
The draw order of the background grid lattice.
|
|
231
|
+
outline_col : `str`, default 'black'
|
|
232
|
+
The edge colour of the per-cell outlines drawn on `indicator == 1`
|
|
233
|
+
cells.
|
|
234
|
+
outline_linestyle : `str`, default '-'
|
|
235
|
+
The linestyle of the per-cell outlines.
|
|
236
|
+
outline_linewidth : `float`, default 1.5
|
|
237
|
+
The width of the per-cell outlines. Set to 0 to suppress them.
|
|
238
|
+
outline_zorder : `int`, `float`, default `2`
|
|
239
|
+
The draw order of the per-cell outlines.
|
|
240
|
+
frame : `bool`, default `False`
|
|
241
|
+
Whether to plot the spines.
|
|
242
|
+
cbar_bool : `bool`, default `False`
|
|
243
|
+
If `True`, add a colourbar (built from the masked heatmap layer).
|
|
244
|
+
cbar_label : `str`, default ""
|
|
245
|
+
The label for the colourbar.
|
|
246
|
+
ax : `plt.Axes` or `None`, default `None`
|
|
247
|
+
A `matplotlib.axes.Axes` instance to draw on. If `None`, a new figure
|
|
248
|
+
and axes are created.
|
|
249
|
+
figsize : `tuple` [`float`, `float`] or `None`, default `None`
|
|
250
|
+
Figure size in inches (width, height). Ignored if `ax` is provided.
|
|
251
|
+
grid_kw : `dict` [`str`, `any`] or `None`, default `None`
|
|
252
|
+
Additional arguments forwarded to `matplotlib.Axes.grid` for the
|
|
253
|
+
background lattice.
|
|
254
|
+
outline_kw : `dict` [`str`, `any`] or `None`, default `None`
|
|
255
|
+
Additional arguments forwarded to each `matplotlib.patches.Rectangle`
|
|
256
|
+
outline. Outlines default to `clip_on=False` so the borders of cells on
|
|
257
|
+
the matrix boundary are not clipped by the axes edge; pass
|
|
258
|
+
`{'clip_on': True}` to restore clipping.
|
|
259
|
+
cbar_kw : `dict` [`str`, `any`] or `None`, default `None`
|
|
260
|
+
A dictionary with arguments to `matplotlib.Figure.colorbar`.
|
|
261
|
+
background_kw : `dict` [`str`, `any`] or `None`, default `None`,
|
|
262
|
+
A dictionary with arguments to `heatmap.heatmap`.
|
|
263
|
+
**kwargs : `any`,
|
|
264
|
+
All other arguments passed to `masking ax.imshow`.
|
|
265
|
+
|
|
266
|
+
Returns
|
|
267
|
+
-------
|
|
268
|
+
im : `matplotlib.image.AxesImage`
|
|
269
|
+
The masked (foreground) heatmap image object.
|
|
270
|
+
cbar : `matplotlib.colorbar.Colorbar` or `None`
|
|
271
|
+
The colourbar object if `cbar_bool` is `True`, otherwise `None`.
|
|
272
|
+
|
|
273
|
+
Notes
|
|
274
|
+
-----
|
|
275
|
+
The masking is achieved by a separate imshow call setting the cells to
|
|
276
|
+
transparent, revealing the background.
|
|
277
|
+
|
|
278
|
+
The returned `im` mirrors the contract of `heatmap` and can therefore be
|
|
279
|
+
annotated using `annotate_heatmap`.
|
|
280
|
+
"""
|
|
281
|
+
# create an axes if needed
|
|
282
|
+
if ax is None:
|
|
283
|
+
_, ax = plt.subplots(figsize=figsize)
|
|
284
|
+
else:
|
|
285
|
+
f = ax.figure
|
|
286
|
+
# check input types
|
|
287
|
+
is_type(data, (pd.DataFrame, np.ndarray))
|
|
288
|
+
is_type(indicator, (pd.DataFrame, np.ndarray))
|
|
289
|
+
_ = [is_type(k, (dict, type(None))) for k in\
|
|
290
|
+
(grid_kw, cbar_kw, outline_kw, background_kw)]
|
|
291
|
+
# the indicator must match the data shape exactly (full 2D shape, not just
|
|
292
|
+
# the row count)
|
|
293
|
+
if np.shape(data) != np.shape(indicator):
|
|
294
|
+
raise InputValidationError(
|
|
295
|
+
f"`indicator` shape {np.shape(indicator)} does not match `data` "
|
|
296
|
+
f"shape {np.shape(data)}."
|
|
297
|
+
)
|
|
298
|
+
# coerce the data and indicator to numpy arrays
|
|
299
|
+
if isinstance(data, pd.DataFrame):
|
|
300
|
+
matrix = data.copy().to_numpy()
|
|
301
|
+
else:
|
|
302
|
+
matrix = data
|
|
303
|
+
if isinstance(indicator, pd.DataFrame):
|
|
304
|
+
flag = indicator.copy().to_numpy()
|
|
305
|
+
else:
|
|
306
|
+
flag = indicator
|
|
307
|
+
# flag should only contain 0 and 1
|
|
308
|
+
unique_flags = set(np.unique(flag).tolist())
|
|
309
|
+
if not unique_flags.issubset({0, 1}):
|
|
310
|
+
raise InputValidationError(
|
|
311
|
+
f"`indicator` must only contain binary (0/1) values, got "
|
|
312
|
+
f"{sorted(unique_flags)}."
|
|
313
|
+
)
|
|
314
|
+
# setup the kwargs None to dict
|
|
315
|
+
background_kw = background_kw or {}
|
|
316
|
+
# masking_kw = masking_kw or {}
|
|
317
|
+
grid_kw = grid_kw or {}
|
|
318
|
+
cbar_kw = cbar_kw or {}
|
|
319
|
+
outline_kw = outline_kw or {}
|
|
320
|
+
outline_kw = _update_kwargs(update_dict=outline_kw,
|
|
321
|
+
zorder=outline_zorder)
|
|
322
|
+
# the background grid lattice carries the background draw order
|
|
323
|
+
grid_kw = _update_kwargs(update_dict=grid_kw, zorder=background_zorder)
|
|
324
|
+
# ### Layer 1: a single-colour background covering every cell. Reusing
|
|
325
|
+
# `heatmap` keeps a single source of truth for the tick, label, spine and
|
|
326
|
+
# grid (lattice) cosmetics.
|
|
327
|
+
background = np.zeros_like(matrix, dtype=float)
|
|
328
|
+
layer1_kwargs = _update_kwargs(
|
|
329
|
+
update_dict=background_kw,
|
|
330
|
+
data=background, row_labels=row_labels, col_labels=col_labels,
|
|
331
|
+
grid_col=background_gridcol, grid_linestyle=background_linestyle,
|
|
332
|
+
grid_linewidth=background_linewidth, cbar_bool=False, ax=ax,
|
|
333
|
+
grid_kw=grid_kw, cmap=ListedColormap([background_col]),
|
|
334
|
+
)
|
|
335
|
+
heatmap(**layer1_kwargs, )
|
|
336
|
+
# ### Layer 2: the heatmap, masked so only `indicator == 1` cells are drawn.
|
|
337
|
+
masked = np.ma.masked_where(flag == 0, matrix)
|
|
338
|
+
# creating an alpha matrix.
|
|
339
|
+
# user_alpha = masking_kw.pop('alpha', 1.0)
|
|
340
|
+
user_alpha = kwargs.pop('alpha', 1.0)
|
|
341
|
+
alpha = (flag == 1).astype(float) * np.asarray(user_alpha, dtype=float)
|
|
342
|
+
layer2_kwargs = _update_kwargs(
|
|
343
|
+
update_dict=kwargs, alpha=alpha,
|
|
344
|
+
)
|
|
345
|
+
im = ax.imshow(masked, **layer2_kwargs)
|
|
346
|
+
# Create colorbar from the foreground (masked) layer
|
|
347
|
+
if cbar_bool:
|
|
348
|
+
cbar = ax.figure.colorbar(im, ax=ax, **cbar_kw)
|
|
349
|
+
cbar.ax.set_ylabel(cbar_label, rotation=-90, va="bottom")
|
|
350
|
+
else:
|
|
351
|
+
cbar = None
|
|
352
|
+
# ### Outline each `indicator == 1` cell. Zero cells get no patch, so they
|
|
353
|
+
# carry no outline; a zero `outline_linewidth` hides the borders.
|
|
354
|
+
rect_kw = _update_kwargs(update_dict=outline_kw, facecolor='none',
|
|
355
|
+
edgecolor=outline_col,
|
|
356
|
+
linestyle=outline_linestyle,
|
|
357
|
+
linewidth=outline_linewidth,
|
|
358
|
+
clip_on=False,
|
|
359
|
+
)
|
|
360
|
+
rows, cols = np.where(flag == 1)
|
|
361
|
+
# NOTE the 0.5 and 1.0 are imshow fixed convention and should be hardcoded
|
|
362
|
+
for i, j in zip(rows, cols):
|
|
363
|
+
ax.add_patch(Rectangle((j-.5, i-.5), 1, 1, **rect_kw))
|
|
364
|
+
# Show the spines
|
|
365
|
+
if frame:
|
|
366
|
+
for spine in ax.spines.values():
|
|
367
|
+
spine.set_visible(True)
|
|
368
|
+
# return stuff
|
|
369
|
+
return im, cbar
|
|
370
|
+
|
|
165
371
|
# ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
166
372
|
def annotate_heatmap(
|
|
167
373
|
im:plt.Axes.imshow,
|
|
@@ -210,6 +416,10 @@ def annotate_heatmap(
|
|
|
210
416
|
|
|
211
417
|
# mapping data to matrix
|
|
212
418
|
values = im.get_array()
|
|
419
|
+
# masked cells (e.g. from `masked_heatmap`) are not drawn and must not be
|
|
420
|
+
# annotated; `getmaskarray` yields a full boolean mask for masked arrays and
|
|
421
|
+
# an all-False mask for plain arrays, leaving the unmasked path unchanged
|
|
422
|
+
mask = np.ma.getmaskarray(values)
|
|
213
423
|
if data is None:
|
|
214
424
|
matrix = im.get_array()
|
|
215
425
|
elif isinstance(data, pd.DataFrame):
|
|
@@ -239,6 +449,9 @@ def annotate_heatmap(
|
|
|
239
449
|
texts = []
|
|
240
450
|
for i in range(matrix.shape[0]):
|
|
241
451
|
for j in range(matrix.shape[1]):
|
|
452
|
+
# skip masked cells, which carry no drawn value to annotate
|
|
453
|
+
if mask[i, j]:
|
|
454
|
+
continue
|
|
242
455
|
# only run if threshold exists
|
|
243
456
|
if threshold is not None:
|
|
244
457
|
kw.update(color=textcolors[int(abs(values[i, j]) >= threshold)])
|
plot_misc/utils/utils.py
CHANGED
|
@@ -201,12 +201,19 @@ class MatrixHeatmapResults(Results):
|
|
|
201
201
|
(unlogged) point estimates, masked where needed.
|
|
202
202
|
curated_matrix_value : `pd.DataFrame`
|
|
203
203
|
The final heatmap matrix with signed -log10(p-values), possibly NA-masked
|
|
204
|
-
and suitable for plotting (numeric).
|
|
204
|
+
and suitable for plotting (numeric). This is the default colour matrix.
|
|
205
|
+
curated_matrix_value_unsigned_log : `pd.DataFrame`
|
|
206
|
+
The unsigned -log10(p-value) matrix, NA-masked with 0 (numeric).
|
|
207
|
+
curated_matrix_value_raw : `pd.DataFrame`
|
|
208
|
+
The raw (untransformed) p-value matrix, NA-masked with 1 (numeric).
|
|
205
209
|
matrix_point_estimate : `pd.DataFrame`
|
|
206
210
|
A matrix of formatted point estimates as strings, with non-significant
|
|
207
211
|
values masked.
|
|
208
212
|
matrix_pvalue : `pd.DataFrame`
|
|
209
|
-
|
|
213
|
+
The p-value annotation matrix (strings). Its representation follows the
|
|
214
|
+
`annotate` argument of `calc_matrices`: signed -log10(p-values) for
|
|
215
|
+
'pvalues'/'pvalues_signed', unsigned -log10(p-values) for
|
|
216
|
+
'pvalues_unsigned', or the raw p-values for 'pvalues_raw'.
|
|
210
217
|
matrix_star : `pd.DataFrame`
|
|
211
218
|
A matrix showing stars for significant values and empty strings otherwise.
|
|
212
219
|
source_data : `pd.DataFrame`
|
|
@@ -219,6 +226,8 @@ class MatrixHeatmapResults(Results):
|
|
|
219
226
|
UtilsNames.annot_star,
|
|
220
227
|
UtilsNames.annot_pval,
|
|
221
228
|
UtilsNames.annot_effect,
|
|
229
|
+
UtilsNames.value_unsigned_log,
|
|
230
|
+
UtilsNames.value_raw,
|
|
222
231
|
UtilsNames.value_original,
|
|
223
232
|
UtilsNames.value_point,
|
|
224
233
|
UtilsNames.source_data,
|
|
@@ -530,98 +539,123 @@ def _extract(data:pd.DataFrame, exposure_col:str, outcome_col:str,
|
|
|
530
539
|
|
|
531
540
|
# ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
532
541
|
def _format_matrices(effect:pd.DataFrame, pval:pd.DataFrame, sig:float,
|
|
533
|
-
|
|
534
|
-
symbol:str='★'
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
542
|
+
ptrun:Real=16, digits:str='3',
|
|
543
|
+
symbol:str='★',
|
|
544
|
+
pval_mode:Literal['signed_log',
|
|
545
|
+
'unsigned_log',
|
|
546
|
+
'raw']='signed_log',
|
|
547
|
+
) -> tuple[pd.DataFrame,
|
|
548
|
+
pd.DataFrame,
|
|
549
|
+
pd.DataFrame,
|
|
550
|
+
pd.DataFrame,
|
|
551
|
+
pd.DataFrame,
|
|
552
|
+
pd.DataFrame,
|
|
553
|
+
pd.DataFrame,
|
|
554
|
+
]:
|
|
540
555
|
"""
|
|
541
556
|
Format effect and p-value matrices for heatmap visualisation.
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
557
|
+
|
|
558
|
+
P-values are always -log10 transformed. Three numeric p-value tables are
|
|
559
|
+
returned (signed -log10, unsigned -log10, and the raw p-value), together
|
|
560
|
+
with the string annotation matrices used to overlay the heatmap.
|
|
545
561
|
|
|
546
562
|
Parameters
|
|
547
563
|
----------
|
|
548
564
|
effect : `pd.DataFrame`
|
|
549
565
|
Matrix of effect estimates as floats.
|
|
550
566
|
pval : `pd.DataFrame`
|
|
551
|
-
Matrix of p-values as floats.
|
|
567
|
+
Matrix of p-values as floats (in [0, 1]).
|
|
552
568
|
sig : `float`
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
log : `bool`, default is `True`
|
|
556
|
-
should the `pval` matrix be -log10 transformed.
|
|
569
|
+
Significance cut-off expressed as a `-log10` threshold (a cell is
|
|
570
|
+
significant when `-log10(p) >= sig`).
|
|
557
571
|
ptrun : `float` or `int`, default 16
|
|
558
|
-
|
|
572
|
+
P-values smaller than `10^(-ptrun)` are truncated.
|
|
559
573
|
digits : `str`, default `3`
|
|
560
|
-
|
|
574
|
+
Number of decimals the numeric tables and effect strings are rounded
|
|
575
|
+
to (a single integer character).
|
|
561
576
|
symbol : `str`, default `★`
|
|
562
|
-
|
|
577
|
+
The glyph used to flag significant findings in the `star` matrix.
|
|
578
|
+
pval_mode : {'signed_log', 'unsigned_log', 'raw'}, default 'signed_log'
|
|
579
|
+
Representation used for the `pvalstring` annotation matrix only:
|
|
580
|
+
- 'signed_log': signed -log10(p-value).
|
|
581
|
+
- 'unsigned_log': unsigned -log10(p-value).
|
|
582
|
+
- 'raw': the untransformed p-value (in [0, 1]).
|
|
583
|
+
This affects *only* the annotation string; it never changes the numeric
|
|
584
|
+
tables or the significance mask.
|
|
563
585
|
|
|
564
586
|
Returns
|
|
565
587
|
-------
|
|
566
|
-
|
|
567
|
-
Signed p-value matrix (
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
588
|
+
pval_signed : `pd.DataFrame`
|
|
589
|
+
Signed -log10(p-value) matrix (`sign(effect) * -log10 p`). This is the
|
|
590
|
+
default heatmap colour matrix.
|
|
591
|
+
pval_unsigned : `pd.DataFrame`
|
|
592
|
+
Unsigned -log10(p-value) matrix.
|
|
593
|
+
pval_raw : `pd.DataFrame`
|
|
594
|
+
Raw (untransformed) p-value matrix, rounded to `digits`.
|
|
595
|
+
effect : `pd.DataFrame`
|
|
596
|
+
Effect matrix as strings, masked to `'.'` where non-significant.
|
|
597
|
+
star : `pd.DataFrame`
|
|
598
|
+
Star annotation matrix (`symbol` for significant cells).
|
|
599
|
+
pvalstring : `pd.DataFrame`
|
|
600
|
+
P-value annotation matrix (strings), rendered per `pval_mode`.
|
|
601
|
+
effect_float : `pd.DataFrame`
|
|
602
|
+
Unmasked effect matrix (numeric).
|
|
577
603
|
|
|
578
|
-
|
|
604
|
+
Raises
|
|
605
|
+
------
|
|
606
|
+
ValueError
|
|
607
|
+
If `digits` is not a single character, or `pval_mode` is not one of
|
|
608
|
+
the supported values.
|
|
609
|
+
|
|
610
|
+
Notes
|
|
611
|
+
-----
|
|
612
|
+
The numeric matrices are signed by the effect direction (`sign(effect)`) so
|
|
613
|
+
a diverging colour map encodes both significance magnitude and effect
|
|
614
|
+
direction on one scale. The significance mask is unaffected by the choice of
|
|
615
|
+
numeric representation: because `-log10` is monotonic, `p <= alpha` and
|
|
616
|
+
`-log10(p) >= -log10(alpha)` select the same cells.
|
|
617
|
+
"""
|
|
618
|
+
# Validate inputs.
|
|
579
619
|
if len(digits) > 1:
|
|
580
620
|
raise ValueError("`digits` must be interpretable as a single integer, "
|
|
581
621
|
f"got: {digits}.")
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
# rounding
|
|
622
|
+
if pval_mode not in ('signed_log', 'unsigned_log', 'raw'):
|
|
623
|
+
raise ValueError("`pval_mode` must be one of 'signed_log', "
|
|
624
|
+
f"'unsigned_log', 'raw', got: {pval_mode}.")
|
|
625
|
+
# ### Compute the three numeric p-value tables (signed/unsigned -log10, raw).
|
|
626
|
+
ndig = int(float(digits))
|
|
588
627
|
dig = '{:.'+digits+'f}'
|
|
589
|
-
|
|
628
|
+
pval_raw = pval.round(ndig)
|
|
629
|
+
nlog10 = _nlog10_func(pval, ptrun)
|
|
630
|
+
pval_unsigned = nlog10.round(ndig)
|
|
590
631
|
dir = np.sign(effect)
|
|
591
|
-
|
|
632
|
+
pval_signed = dir * pval_unsigned
|
|
633
|
+
# ### Keep the unmasked effect matrix and format the effect estimates.
|
|
592
634
|
effect_float = effect.copy()
|
|
593
|
-
# formatting
|
|
594
635
|
if pd.__version__ < '2.1.0':
|
|
595
636
|
effect = effect.applymap(dig.format).copy()
|
|
596
637
|
else:
|
|
597
638
|
effect = effect.map(dig.format).copy()
|
|
598
|
-
#
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
star = effect.copy()
|
|
607
|
-
star[pval_full >= sig] = symbol
|
|
608
|
-
# pvalues
|
|
609
|
-
pvalstring = effect.copy()
|
|
610
|
-
pvalstring[pval_full >= sig] = pval[pval_full >= sig].astype('str')
|
|
611
|
-
# if log != True use smaller than
|
|
639
|
+
# Derive the significance mask (-log10 space; NaN cells fall through both).
|
|
640
|
+
significant = nlog10 >= sig
|
|
641
|
+
not_significant = nlog10 < sig
|
|
642
|
+
# #### Select the annotation representation (`pval_mode` axis).
|
|
643
|
+
if pval_mode == 'raw':
|
|
644
|
+
pval_annot = pval_raw
|
|
645
|
+
elif pval_mode == 'unsigned_log':
|
|
646
|
+
pval_annot = pval_unsigned
|
|
612
647
|
else:
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
return pval, effect, star, pvalstring, effect_float
|
|
648
|
+
pval_annot = pval_signed
|
|
649
|
+
# #### Mask non-significant cells and build the star / p-value annotations.
|
|
650
|
+
effect[not_significant] = '.'
|
|
651
|
+
effect = effect.astype('str')
|
|
652
|
+
star = effect.copy()
|
|
653
|
+
star[significant] = symbol
|
|
654
|
+
pvalstring = effect.copy()
|
|
655
|
+
pvalstring[significant] = pval_annot[significant].astype('str')
|
|
656
|
+
# Return the tables
|
|
657
|
+
return (pval_signed, pval_unsigned, pval_raw, effect, star, pvalstring,
|
|
658
|
+
effect_float)
|
|
625
659
|
|
|
626
660
|
# ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
627
661
|
def calc_matrices(data:pd.DataFrame,
|
|
@@ -629,10 +663,17 @@ def calc_matrices(data:pd.DataFrame,
|
|
|
629
663
|
outcome_col:str,
|
|
630
664
|
point_col:str='point',
|
|
631
665
|
pvalue_col:str='pvalue',
|
|
632
|
-
alpha:Real
|
|
666
|
+
alpha:Real=0.05,
|
|
633
667
|
sig_numbers:int=2,
|
|
634
|
-
ptrun:Real=16,
|
|
635
|
-
annotate:
|
|
668
|
+
ptrun:Real=1e-16,
|
|
669
|
+
annotate:Literal['symbol',
|
|
670
|
+
'star',
|
|
671
|
+
'pvalues',
|
|
672
|
+
'pvalues_signed',
|
|
673
|
+
'pvalues_unsigned',
|
|
674
|
+
'pvalues_raw',
|
|
675
|
+
'point_estimates']|None='symbol',
|
|
676
|
+
symbol:str='★',
|
|
636
677
|
without_log:bool=False,
|
|
637
678
|
mask_na:bool=True,
|
|
638
679
|
**kwargs:Any,
|
|
@@ -659,20 +700,35 @@ def calc_matrices(data:pd.DataFrame,
|
|
|
659
700
|
pvalue_col : `str`, default 'pvalue'
|
|
660
701
|
Column name with p-values. Note p-values are expected to range between
|
|
661
702
|
0 and 1.
|
|
662
|
-
alpha : `float
|
|
663
|
-
The significance cut-off
|
|
703
|
+
alpha : `float`, default `0.05`
|
|
704
|
+
The significance cut-off as a raw p-value in (0, 1] (consistent with
|
|
705
|
+
`volcano`). Converted internally to a -log10 threshold. Values outside
|
|
706
|
+
(0, 1] raise `InputValidationError`.
|
|
664
707
|
sig_numbers : `int`, default 2
|
|
665
708
|
The number of significant numbers the cell annotations should have.
|
|
666
|
-
ptrun : `float` or `int`, default 16
|
|
667
|
-
|
|
668
|
-
|
|
709
|
+
ptrun : `float` or `int`, default 1e-16
|
|
710
|
+
The truncation threshold as a raw p-value in (0, 1]: p-values smaller
|
|
711
|
+
than `ptrun` are floored to `ptrun` before the -log10 transform.
|
|
712
|
+
annotate : `str`, default 'symbol'
|
|
669
713
|
Annotation style to return. Options:
|
|
670
|
-
- '
|
|
671
|
-
- '
|
|
672
|
-
- '
|
|
714
|
+
- 'symbol': significance markers using `symbol`.
|
|
715
|
+
- 'star': **Deprecated** alias for 'symbol' — use 'symbol' instead.
|
|
716
|
+
- 'pvalues': signed -log10(p-values). **Deprecated** — use
|
|
717
|
+
'pvalues_signed' instead.
|
|
718
|
+
- 'pvalues_signed': signed -log10(p-values).
|
|
719
|
+
- 'pvalues_unsigned': unsigned -log10(p-values).
|
|
720
|
+
- 'pvalues_raw': the untransformed p-values (in [0, 1]).
|
|
721
|
+
- 'point_estimates': formatted effect estimates
|
|
673
722
|
- None: returns only numeric matrix without annotations
|
|
723
|
+
The numeric value matrix is always signed -log10(p-values); only the
|
|
724
|
+
annotation representation changes with the 'pvalues*' options.
|
|
725
|
+
symbol : `str`, default `★`
|
|
726
|
+
The text/unicode glyph used to flag significant findings when
|
|
727
|
+
`annotate='symbol'` (e.g. `'●'`, `'◆'`, `'*'`).
|
|
674
728
|
without_log : `bool`, default `False`
|
|
675
|
-
|
|
729
|
+
**Deprecated** and no longer changes behaviour: p-values are always
|
|
730
|
+
-log10 transformed. Setting it to `True` emits a `DeprecationWarning`.
|
|
731
|
+
Use the `curated_matrix_value_raw` table for raw p-values.
|
|
676
732
|
mask_na : `bool`, default `True`
|
|
677
733
|
If you want to mask missing results (e.g., replacing NAs by 0 or 1)
|
|
678
734
|
**kwargs
|
|
@@ -686,6 +742,8 @@ def calc_matrices(data:pd.DataFrame,
|
|
|
686
742
|
------
|
|
687
743
|
ValueError
|
|
688
744
|
If `annotate` is not one of the supported values.
|
|
745
|
+
InputValidationError
|
|
746
|
+
If `alpha` is not a raw p-value in (0, 1].
|
|
689
747
|
"""
|
|
690
748
|
#### check input
|
|
691
749
|
is_type(data, pd.DataFrame)
|
|
@@ -694,9 +752,51 @@ def calc_matrices(data:pd.DataFrame,
|
|
|
694
752
|
is_type(point_col, str)
|
|
695
753
|
is_type(pvalue_col, str)
|
|
696
754
|
is_type(alpha, (int, float))
|
|
755
|
+
is_type(ptrun, (int, float))
|
|
697
756
|
is_type(sig_numbers, int)
|
|
757
|
+
is_type(symbol, str)
|
|
698
758
|
is_type(without_log, bool)
|
|
699
759
|
is_type(mask_na, bool)
|
|
760
|
+
### `alpha` is a raw p-value threshold in (0, 1]
|
|
761
|
+
if not (0 < alpha <= 1):
|
|
762
|
+
raise InputValidationError(
|
|
763
|
+
"`alpha` must be a raw p-value in (0, 1] (e.g. 0.05); got: "
|
|
764
|
+
f"{alpha}. The -log10 convention was removed in v2.3."
|
|
765
|
+
)
|
|
766
|
+
### `ptrun` is a raw p-value truncation threshold in (0, 1]
|
|
767
|
+
if not (0 < ptrun <= 1):
|
|
768
|
+
raise InputValidationError(
|
|
769
|
+
"`ptrun` must be a raw p-value in (0, 1] (e.g. 1e-16); got: "
|
|
770
|
+
f"{ptrun}. The exponent convention was removed in v2.3."
|
|
771
|
+
)
|
|
772
|
+
### `without_log` is deprecated and no longer changes behaviour
|
|
773
|
+
if without_log:
|
|
774
|
+
warnings.warn(
|
|
775
|
+
"`without_log` is deprecated and will be removed in a future "
|
|
776
|
+
"release; p-values are now always -log10 transformed. Use the "
|
|
777
|
+
"`curated_matrix_value_raw` table for raw p-values.",
|
|
778
|
+
DeprecationWarning,
|
|
779
|
+
stacklevel=2,
|
|
780
|
+
)
|
|
781
|
+
### warn on deprecated annotation aliases
|
|
782
|
+
deprecated_annot = {
|
|
783
|
+
UtilsNames.mat_annot_star: UtilsNames.mat_annot_symbol,
|
|
784
|
+
UtilsNames.mat_annot_pval: UtilsNames.mat_annot_pval_signed,
|
|
785
|
+
}
|
|
786
|
+
if annotate in deprecated_annot:
|
|
787
|
+
warnings.warn(
|
|
788
|
+
f"`annotate={annotate!r}` is deprecated and will be removed in a "
|
|
789
|
+
f"future release; use {deprecated_annot[annotate]!r} instead.",
|
|
790
|
+
DeprecationWarning,
|
|
791
|
+
stacklevel=2,
|
|
792
|
+
)
|
|
793
|
+
### map the requested annotation onto a p-value representation mode
|
|
794
|
+
if annotate == UtilsNames.mat_annot_pval_unsigned:
|
|
795
|
+
pval_mode = 'unsigned_log'
|
|
796
|
+
elif annotate == UtilsNames.mat_annot_pval_raw:
|
|
797
|
+
pval_mode = 'raw'
|
|
798
|
+
else:
|
|
799
|
+
pval_mode = 'signed_log'
|
|
700
800
|
### subsetting data
|
|
701
801
|
point_mat, pvalue_mat = _extract(data,
|
|
702
802
|
exposure_col=exposure_col,
|
|
@@ -705,17 +805,22 @@ def calc_matrices(data:pd.DataFrame,
|
|
|
705
805
|
pvalue_col=pvalue_col,
|
|
706
806
|
**kwargs,
|
|
707
807
|
)
|
|
708
|
-
### formatting data
|
|
709
|
-
|
|
808
|
+
### formatting data (convert the raw-p `alpha` to a -log10 threshold)
|
|
809
|
+
sig = -1 * np.log10(alpha)
|
|
810
|
+
(values, values_unsigned, values_raw, annot_effect, annot_star,
|
|
811
|
+
annot_pval, values_point) =\
|
|
710
812
|
_format_matrices(
|
|
711
|
-
point_mat, pvalue_mat, sig=
|
|
712
|
-
ptrun
|
|
713
|
-
|
|
813
|
+
point_mat, pvalue_mat, sig=sig,
|
|
814
|
+
ptrun=-np.log10(ptrun), digits=str(sig_numbers),
|
|
815
|
+
symbol=symbol, pval_mode=pval_mode,
|
|
714
816
|
)
|
|
715
817
|
### selecting the annotation to use
|
|
716
|
-
if annotate
|
|
818
|
+
if annotate in (UtilsNames.mat_annot_symbol, UtilsNames.mat_annot_star):
|
|
717
819
|
annot = annot_star
|
|
718
|
-
elif annotate
|
|
820
|
+
elif annotate in (UtilsNames.mat_annot_pval,
|
|
821
|
+
UtilsNames.mat_annot_pval_signed,
|
|
822
|
+
UtilsNames.mat_annot_pval_unsigned,
|
|
823
|
+
UtilsNames.mat_annot_pval_raw):
|
|
719
824
|
annot = annot_pval
|
|
720
825
|
elif annotate == UtilsNames.mat_annot_point:
|
|
721
826
|
annot = annot_effect
|
|
@@ -725,26 +830,30 @@ def calc_matrices(data:pd.DataFrame,
|
|
|
725
830
|
else:
|
|
726
831
|
raise ValueError('Incorrect `annotate` value supplied '
|
|
727
832
|
'Please use: {}'.\
|
|
728
|
-
format([UtilsNames.
|
|
833
|
+
format([UtilsNames.mat_annot_symbol,
|
|
834
|
+
UtilsNames.mat_annot_star,
|
|
729
835
|
UtilsNames.mat_annot_pval,
|
|
836
|
+
UtilsNames.mat_annot_pval_signed,
|
|
837
|
+
UtilsNames.mat_annot_pval_unsigned,
|
|
838
|
+
UtilsNames.mat_annot_pval_raw,
|
|
730
839
|
UtilsNames.mat_annot_point,
|
|
731
840
|
UtilsNames.mat_annot_none,
|
|
732
841
|
]
|
|
733
842
|
))
|
|
734
843
|
### drop or mask NAs
|
|
735
844
|
if not mask_na:
|
|
845
|
+
# drop rows/columns containing any missing value
|
|
736
846
|
drop_c = ~values.isna().any(axis=0)
|
|
737
847
|
drop_r = ~values.isna().any(axis=1)
|
|
738
848
|
values_input = values.loc[drop_r, drop_c]
|
|
849
|
+
values_unsigned_input = values_unsigned.loc[drop_r, drop_c]
|
|
850
|
+
values_raw_input = values_raw.loc[drop_r, drop_c]
|
|
739
851
|
annot_input = annot.loc[drop_r, drop_c]
|
|
740
|
-
# Mask with zero if logged
|
|
741
|
-
elif not without_log:
|
|
742
|
-
values_input = values.fillna(0, inplace=False)
|
|
743
|
-
annot_input = annot.fillna('.', inplace=False)
|
|
744
|
-
annot_input[annot_input == 'nan'] = '.'
|
|
745
|
-
# Mask with one if not
|
|
746
852
|
else:
|
|
747
|
-
|
|
853
|
+
# fill missing -log10 values with 0 and raw p-values with 1
|
|
854
|
+
values_input = values.fillna(0, inplace=False)
|
|
855
|
+
values_unsigned_input = values_unsigned.fillna(0, inplace=False)
|
|
856
|
+
values_raw_input = values_raw.fillna(1, inplace=False)
|
|
748
857
|
annot_input = annot.fillna('.', inplace=False)
|
|
749
858
|
annot_input[annot_input == 'nan'] = '.'
|
|
750
859
|
### Return
|
|
@@ -753,6 +862,8 @@ def calc_matrices(data:pd.DataFrame,
|
|
|
753
862
|
UtilsNames.annot_star: annot_star,
|
|
754
863
|
UtilsNames.annot_pval: annot_pval,
|
|
755
864
|
UtilsNames.annot_effect: annot_effect,
|
|
865
|
+
UtilsNames.value_unsigned_log: values_unsigned_input,
|
|
866
|
+
UtilsNames.value_raw: values_raw_input,
|
|
756
867
|
UtilsNames.value_original: values,
|
|
757
868
|
UtilsNames.value_point: values_point,
|
|
758
869
|
UtilsNames.source_data: data,
|
|
@@ -979,7 +1090,7 @@ def segment_labelled(
|
|
|
979
1090
|
# do we need to apply a transformation first
|
|
980
1091
|
if calc_angle_after_trans:
|
|
981
1092
|
p1 = list(ax.transData.transform_point((x[0], y[0])))
|
|
982
|
-
p2 = list(ax.transData.transform_point((
|
|
1093
|
+
p2 = list(ax.transData.transform_point((x[1], y[1])))
|
|
983
1094
|
x_trans=[p1[0], p2[0]]
|
|
984
1095
|
y_trans=[p1[1], p2[1]]
|
|
985
1096
|
else:
|
plot_misc/volcano.py
CHANGED
|
@@ -39,8 +39,7 @@ from typing import Any
|
|
|
39
39
|
|
|
40
40
|
# ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
|
41
41
|
def plot_volcano(data:DataFrame, y_column:str, x_column:str,
|
|
42
|
-
point_label:str | None = None,
|
|
43
|
-
fsize:tuple[float,float] | None = None, adjust:bool=False,
|
|
42
|
+
point_label:str | None = None, adjust:bool=False,
|
|
44
43
|
lim:int=1000, vline:Real=0, alpha:float=1e-5,
|
|
45
44
|
col_sgnd: str = 'orangered', col_nsgnd: str = 'dimgrey',
|
|
46
45
|
col_vline: str = 'lightcoral',
|
|
@@ -50,6 +49,7 @@ def plot_volcano(data:DataFrame, y_column:str, x_column:str,
|
|
|
50
49
|
index_label:list[str] | None = None,
|
|
51
50
|
font_label: str | None = None,
|
|
52
51
|
ax:plt.Axes | None = None,
|
|
52
|
+
figsize:tuple[float,float] | None = None,
|
|
53
53
|
label_kwargs_dict:dict[Any,Any] | None = None,
|
|
54
54
|
scatter_sig_kwargs_dict:dict[Any,Any] | None = None,
|
|
55
55
|
scatter_nonsig_kwargs_dict:dict[Any,Any] | None = None,
|
|
@@ -70,8 +70,6 @@ def plot_volcano(data:DataFrame, y_column:str, x_column:str,
|
|
|
70
70
|
point_label : `str` or `None`, default `None`
|
|
71
71
|
Column name in `data` to use for point labels. If `None`, no labels
|
|
72
72
|
are added.
|
|
73
|
-
fsize : `tuple` [`float`, `float`] or `None`, default `None`
|
|
74
|
-
Figure size in inches (width, height). Ignored if `ax` is provided.
|
|
75
73
|
adjust : `bool`, default `False`
|
|
76
74
|
Whether to apply label de-overlapping using `adjustText`.
|
|
77
75
|
lim : `int`, default 1000
|
|
@@ -106,6 +104,8 @@ def plot_volcano(data:DataFrame, y_column:str, x_column:str,
|
|
|
106
104
|
Font family to use for point labels (e.g. 'monospace', 'Arial').
|
|
107
105
|
ax : `plt.axes` or `None`, default `None`
|
|
108
106
|
Axis object to plot on. If `None`, a new figure and axis are created.
|
|
107
|
+
figsize : `tuple` [`float`, `float`] or `None`, default `None`
|
|
108
|
+
Figure size in inches (width, height). Ignored if `ax` is provided.
|
|
109
109
|
label_kwargs_dict : `dict` or `None`, default `None`
|
|
110
110
|
Optional keyword arguments passed to `adjust_text`.
|
|
111
111
|
scatter_sig_kwargs_dict : `dict` or `None`, default `None`
|
|
@@ -155,7 +155,7 @@ def plot_volcano(data:DataFrame, y_column:str, x_column:str,
|
|
|
155
155
|
### getting figure
|
|
156
156
|
# should we create a figure and axis
|
|
157
157
|
if ax is None:
|
|
158
|
-
f, ax = plt.subplots(figsize=
|
|
158
|
+
f, ax = plt.subplots(figsize=figsize)
|
|
159
159
|
else:
|
|
160
160
|
f = ax.figure
|
|
161
161
|
### significance level
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: plot-misc
|
|
3
|
-
Version: 2.2.
|
|
3
|
+
Version: 2.2.2
|
|
4
4
|
Summary: Various plotting templates built on top of matplotlib
|
|
5
5
|
Author-email: A Floriaan Schmidt <floriaanschmidt@gmail.com>
|
|
6
6
|
License-Expression: GPL-3.0-or-later
|
|
@@ -48,15 +48,29 @@ Dynamic: license-file
|
|
|
48
48
|
<img src="https://schmidtaf.gitlab.io/plot-misc/_images/icon.png" alt="plot-misc icon" width="250"/>
|
|
49
49
|
|
|
50
50
|
# A collection of plotting functions
|
|
51
|
-
__version__: `2.2.
|
|
51
|
+
__version__: `2.2.2`
|
|
52
52
|
|
|
53
53
|
This repository collects plotting modules written on top of `matplotlib`.
|
|
54
|
-
The functions
|
|
55
|
-
can be customised using the standard matplotlib interface
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
54
|
+
The functions describe plotting archetypes intended to set up light-touch,
|
|
55
|
+
illustrations that can be customised using the standard matplotlib interface
|
|
56
|
+
via axes and figures.
|
|
57
|
+
Because the implementation is matplotlib-first, the API is consistent with
|
|
58
|
+
matplotlib conventions, and users already familiar with the library will find
|
|
59
|
+
the learning curve minimal.
|
|
60
|
+
|
|
61
|
+
The functionality is geared towards illustrations commonly used in biomedical
|
|
62
|
+
research:
|
|
63
|
+
|
|
64
|
+
* Bar charts
|
|
65
|
+
* Bubble charts
|
|
66
|
+
* Forest plots (with optional side-tables)
|
|
67
|
+
* Heatmaps (with optional annotations)
|
|
68
|
+
* Incidence matrix plots
|
|
69
|
+
* Machine learning plots (calibration, feature importance, net benefit)
|
|
70
|
+
* Pie charts
|
|
71
|
+
* Survival plots (with optional survival table)
|
|
72
|
+
* Tree/compatibility plots
|
|
73
|
+
* Volcano plots
|
|
60
74
|
|
|
61
75
|
Please consult the **[documentation](https://SchmidtAF.gitlab.io/plot-misc/)**
|
|
62
76
|
for plot-misc.
|
|
@@ -1,17 +1,17 @@
|
|
|
1
1
|
plot_misc/__init__.py,sha256=ZygAIkX6Nbjag1czWdQa-yP-GM1mBE_9ss21Xh__JFc,34
|
|
2
|
-
plot_misc/_version.py,sha256
|
|
2
|
+
plot_misc/_version.py,sha256=jsJ9CNIuUt8dDFB4i0PiBf07nzBU0RtG1CVRQ7TdoQ0,22
|
|
3
3
|
plot_misc/barchart.py,sha256=_4Fj8ypYi6Aehgv4b13Lh1zS1ee4z7OXrGo1P5QBshw,21925
|
|
4
|
-
plot_misc/constants.py,sha256=
|
|
4
|
+
plot_misc/constants.py,sha256=SABaNQpTRXm4pMF_F6sdsBUvNVyb-WXQLT_YAmUgtuo,4530
|
|
5
5
|
plot_misc/errors.py,sha256=ZbOZg1TbY-7dtPAsJNxIRsjlUfJg0vrx8kRPFsMClaQ,9637
|
|
6
6
|
plot_misc/forest.py,sha256=suIeIY-eGHjzrEb3uargKXLXbSR4kIJ3jhdslaE977I,62508
|
|
7
|
-
plot_misc/heatmap.py,sha256=
|
|
7
|
+
plot_misc/heatmap.py,sha256=mON3VNdYZOKsUWARPapepcz3G8_6iCp9_9ZChx2Wcgc,19852
|
|
8
8
|
plot_misc/incidencematrix.py,sha256=BFDLQZT0nEhd8XzCtVVo7mR3s70DwzAMG4-ut7zLal8,18832
|
|
9
9
|
plot_misc/machine_learning.py,sha256=R60w12PxLnEWQ9CNwgQKTF3S1K9JXkC68TgAzr-ib2s,48631
|
|
10
10
|
plot_misc/piechart.py,sha256=nKT9Xr-rP9t8hv6ITb_0594AlzgfnAEdqkyCslLriLc,8613
|
|
11
11
|
plot_misc/survival.py,sha256=-C3s54qfxmIIaMp8ertaNjEiOMraoe0ZpP_5wYUqgKs,25405
|
|
12
|
-
plot_misc/volcano.py,sha256=
|
|
12
|
+
plot_misc/volcano.py,sha256=gEUOBYUqOE6fzjrS-2asgxi30EPv_Qjfk23lwCx2Cv4,9347
|
|
13
13
|
plot_misc/example_data/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
|
|
14
|
-
plot_misc/example_data/examples.py,sha256=
|
|
14
|
+
plot_misc/example_data/examples.py,sha256=T9jBdtQEbZKEWVoT0SJt8sJ5fTSdXpTqfDA6Y898s0g,31498
|
|
15
15
|
plot_misc/example_data/example_datasets/bar_points.tsv.gz,sha256=ppYQY01DXz77GzSANpwZ9N4zZvxQN5AiHmLsOhZ6cbk,255
|
|
16
16
|
plot_misc/example_data/example_datasets/barchart.tsv.gz,sha256=9TVBaC7yuDuNzOQivTm-UxjmB_jyCfbv1twiL13WpOs,123
|
|
17
17
|
plot_misc/example_data/example_datasets/calibration_bins.tsv.gz,sha256=7BQdx7Tx5zF6wSyyCTwjkSaELKywoykUpelWvxe5pxA,332
|
|
@@ -27,9 +27,9 @@ plot_misc/example_data/example_datasets/volcano.tsv.gz,sha256=Cd0J-VeiVPruUqdbPo
|
|
|
27
27
|
plot_misc/utils/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
|
|
28
28
|
plot_misc/utils/colour.py,sha256=SfhjFcnvTQ-_zfstMqb0vMPyW3-pdPO4G3KJ7-BtQ4E,7715
|
|
29
29
|
plot_misc/utils/formatting.py,sha256=1sGd3d1C5UCtYkeK6RNfJf9joaGbBOfJ8y0lFT5sXD8,14023
|
|
30
|
-
plot_misc/utils/utils.py,sha256=
|
|
31
|
-
plot_misc-2.2.
|
|
32
|
-
plot_misc-2.2.
|
|
33
|
-
plot_misc-2.2.
|
|
34
|
-
plot_misc-2.2.
|
|
35
|
-
plot_misc-2.2.
|
|
30
|
+
plot_misc/utils/utils.py,sha256=5q9-h9OowWVSyas-GS2NSKTakH0kE0A5GM8nfXt6qiA,50779
|
|
31
|
+
plot_misc-2.2.2.dist-info/licenses/LICENSE,sha256=TAKLZop-WRQ03q1liEZVOMPZOrVwR2ndHSdxC7j1YeU,779
|
|
32
|
+
plot_misc-2.2.2.dist-info/METADATA,sha256=Zd_W6_Cc1ZOzGFAe-L5yc4MKUkqvl3GUhX-3DQ-9oDs,5700
|
|
33
|
+
plot_misc-2.2.2.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
|
|
34
|
+
plot_misc-2.2.2.dist-info/top_level.txt,sha256=WyOYx7sloAXvlDvWfFf57ZAf9PnKTWb0zqenkcZKkAE,10
|
|
35
|
+
plot_misc-2.2.2.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|