vima-spatial 0.2.7__tar.gz → 0.2.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/PKG-INFO +1 -1
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/setup.cfg +1 -1
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/ingest/dimreduce.py +108 -49
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/ingest/ingest.py +157 -1
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/ingest/nonst.py +9 -5
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/ingest/st.py +128 -114
- vima_spatial-0.2.9/src/vima/ingest/util.py +237 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima_spatial.egg-info/PKG-INFO +1 -1
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima_spatial.egg-info/SOURCES.txt +3 -1
- vima_spatial-0.2.9/tests/test_collapse_markers.py +155 -0
- vima_spatial-0.2.9/tests/test_st_rasterize.py +363 -0
- vima_spatial-0.2.7/src/vima/ingest/util.py +0 -108
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/README.md +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/pyproject.toml +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/__init__.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/_settings.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/cc.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/data/__init__.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/data/download.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/data/patchcollection.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/data/samples.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/fingerprints.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/ingest/__init__.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/__init__.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/resnet_vae.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/resnetlight_decoder.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/resnetlight_encoder.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/simple_vae.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/vae.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/patchfeatures.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/train/__init__.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/train/logging.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/train/training.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/__init__.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/features.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/patches.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/patchexamples.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/spatial.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/umaps.py +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima_spatial.egg-info/dependency_links.txt +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima_spatial.egg-info/requires.txt +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima_spatial.egg-info/top_level.txt +0 -0
- {vima_spatial-0.2.7 → vima_spatial-0.2.9}/tests/test_ra_regression.py +0 -0
|
@@ -16,10 +16,10 @@ def metapixels_allsamples(normedpixelsdir, masksdir, sids, total_n_metapixels):
|
|
|
16
16
|
"""
|
|
17
17
|
Pool metapixels across all samples for a more robust PCA fit.
|
|
18
18
|
|
|
19
|
-
Loads each sample's normalized pixels,
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
19
|
+
Loads each sample's normalized pixels, builds roughly
|
|
20
|
+
``total_n_metapixels // len(sids)`` randomly chosen metapixels from it, and
|
|
21
|
+
standardizes them with the stored per-marker means/stds. Warns if a sample's
|
|
22
|
+
markers differ from the first sample's.
|
|
23
23
|
|
|
24
24
|
Parameters
|
|
25
25
|
----------
|
|
@@ -55,7 +55,7 @@ def metapixels_allsamples(normedpixelsdir, masksdir, sids, total_n_metapixels):
|
|
|
55
55
|
for i, sid in enumerate(settings.progress(sids, name='creating metapixels')):
|
|
56
56
|
da = xr.open_dataarray(f'{normedpixelsdir}/{sid}.nc')
|
|
57
57
|
mask_da = xr.open_dataarray(f'{masksdir}/{sid}.nc')
|
|
58
|
-
|
|
58
|
+
|
|
59
59
|
# ensure same markers in same order in all files
|
|
60
60
|
markers = list(da.marker.values)
|
|
61
61
|
if ref_markers is None:
|
|
@@ -68,18 +68,21 @@ def metapixels_allsamples(normedpixelsdir, masksdir, sids, total_n_metapixels):
|
|
|
68
68
|
f'{len(missing)} missing, {len(extra)} extra vs {ref_sid}')
|
|
69
69
|
logger.warning(f'{sid} has different markers ({len(markers)}) '
|
|
70
70
|
f'than {ref_sid} ({len(ref_markers)}): {detail}')
|
|
71
|
-
|
|
72
|
-
means = xr.DataArray(da.attrs['means'], dims='marker')
|
|
73
|
-
stds = xr.DataArray(da.attrs['stds'], dims='marker')
|
|
74
|
-
da = ((da - means) / stds).where(mask_da, 0)
|
|
75
71
|
|
|
76
|
-
|
|
72
|
+
means, stds = da.attrs['means'], da.attrs['stds']
|
|
73
|
+
|
|
74
|
+
# metapixels are built from the un-standardized pixels and standardized
|
|
75
|
+
# afterward: averaging over a window and the affine (x-mean)/std commute,
|
|
76
|
+
# so this is identical to standardizing the full array first but avoids
|
|
77
|
+
# materializing several (y, x, marker)-sized temporaries.
|
|
78
|
+
mp, all_npixels[sid] = metapixels(da, mask_da, n_metapixels=nmp_per_sample)
|
|
77
79
|
da.close(); mask_da.close()
|
|
78
80
|
del da, mask_da
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
81
|
+
|
|
82
|
+
mp -= means
|
|
83
|
+
mp /= stds
|
|
84
|
+
all_metapixels[sid] = pd.DataFrame(data=mp, columns=markers)
|
|
85
|
+
del mp
|
|
83
86
|
|
|
84
87
|
# visualize distribution of num non-empty pixels per metapixel in this sample
|
|
85
88
|
if settings.show_plots():
|
|
@@ -95,33 +98,61 @@ def metapixels_allsamples(normedpixelsdir, masksdir, sids, total_n_metapixels):
|
|
|
95
98
|
|
|
96
99
|
return all_metapixels, all_npixels
|
|
97
100
|
|
|
98
|
-
def metapixels(s, mask, npixels_thresh=0):
|
|
101
|
+
def metapixels(s, mask, npixels_thresh=0, n_metapixels=None, window=5):
|
|
99
102
|
"""
|
|
100
103
|
Pool each pixel with its neighbors into a metapixel.
|
|
101
104
|
|
|
102
|
-
|
|
103
|
-
|
|
105
|
+
Averages each marker over a ``window``-by-``window`` window centered on a
|
|
106
|
+
pixel, using only the non-empty (masked) pixels in that window, so each
|
|
104
107
|
metapixel is the average over its non-empty neighbors. Metapixels with at
|
|
105
108
|
most ``npixels_thresh`` contributing pixels are dropped.
|
|
106
109
|
|
|
110
|
+
Parameters
|
|
111
|
+
----------
|
|
112
|
+
n_metapixels
|
|
113
|
+
If given, uniformly sample at most this many metapixel centers and
|
|
114
|
+
compute only those. Since the caller typically keeps a small random
|
|
115
|
+
subset anyway, this avoids convolving the whole (y, x, marker) array,
|
|
116
|
+
which dominates the cost for large marker panels.
|
|
117
|
+
|
|
107
118
|
Returns
|
|
108
119
|
-------
|
|
109
120
|
tuple
|
|
110
|
-
``(
|
|
111
|
-
|
|
121
|
+
``(metapixels, npixels)``: a ``(metapixel, marker)`` float32 array and
|
|
122
|
+
the per-metapixel count of contributing non-empty pixels.
|
|
112
123
|
"""
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
npixels = convolve(mask.
|
|
119
|
-
|
|
120
|
-
#
|
|
121
|
-
|
|
124
|
+
mask = mask.data
|
|
125
|
+
H, W = mask.shape
|
|
126
|
+
|
|
127
|
+
# how many non-empty pixels contribute to each candidate metapixel (cheap: 2D only)
|
|
128
|
+
kernel = np.ones((window, window), np.float32)
|
|
129
|
+
npixels = convolve(mask.astype(np.float32), kernel, mode="constant")
|
|
130
|
+
|
|
131
|
+
# pick the metapixel centers, sampling before doing any work over markers
|
|
132
|
+
centers = np.flatnonzero(npixels.ravel() > npixels_thresh)
|
|
133
|
+
if n_metapixels is not None and len(centers) > n_metapixels:
|
|
134
|
+
centers = centers[np.random.choice(len(centers), n_metapixels, replace=False)]
|
|
135
|
+
npixels = npixels.ravel()[centers]
|
|
136
|
+
|
|
137
|
+
# sum each window by gathering its non-empty pixels, one neighbor offset at a time
|
|
138
|
+
data = s.data.reshape(H * W, -1)
|
|
139
|
+
mask = mask.ravel()
|
|
140
|
+
r, c = np.divmod(centers, W)
|
|
141
|
+
mp = np.zeros((len(centers), data.shape[1]), np.float32)
|
|
142
|
+
rad = window // 2
|
|
143
|
+
for dr in range(-rad, rad + 1):
|
|
144
|
+
rr = r + dr
|
|
145
|
+
for dc in range(-rad, rad + 1):
|
|
146
|
+
cc = c + dc
|
|
147
|
+
neighbor = rr * W + cc
|
|
148
|
+
contributes = (rr >= 0) & (rr < H) & (cc >= 0) & (cc < W)
|
|
149
|
+
contributes &= mask[np.where(contributes, neighbor, 0)]
|
|
150
|
+
i = np.flatnonzero(contributes)
|
|
151
|
+
mp[i] += data[neighbor[i]]
|
|
122
152
|
|
|
123
153
|
# divide each metapixel by the # of non-empty pixels that contributed to it and return
|
|
124
|
-
|
|
154
|
+
mp /= npixels[:, None]
|
|
155
|
+
return mp, npixels
|
|
125
156
|
|
|
126
157
|
# mps should be an array of dataframes containing metapixels
|
|
127
158
|
def pca_metapixels(mps, k):
|
|
@@ -142,17 +173,33 @@ def pca_metapixels(mps, k):
|
|
|
142
173
|
Returns
|
|
143
174
|
-------
|
|
144
175
|
tuple
|
|
145
|
-
``(loadings,
|
|
146
|
-
|
|
176
|
+
``(loadings, allmp)``: the gene-by-component loading matrix and the
|
|
177
|
+
standardized metapixel AnnData.
|
|
147
178
|
"""
|
|
148
179
|
logger.info('merging and standardizing metapixels')
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
180
|
+
mps = list(mps)
|
|
181
|
+
markers = mps[0].columns
|
|
182
|
+
n = sum(len(mp) for mp in mps)
|
|
183
|
+
|
|
184
|
+
# merge into one preallocated float32 matrix and standardize it in place,
|
|
185
|
+
# accumulating the moments in float64. Doing this in pandas instead upcasts
|
|
186
|
+
# the whole matrix to float64 and copies it once per operation, which at
|
|
187
|
+
# these sizes costs more than the PCA itself.
|
|
188
|
+
allmp = np.empty((n, len(markers)), np.float32)
|
|
189
|
+
i = 0
|
|
190
|
+
for mp in mps:
|
|
191
|
+
allmp[i:i+len(mp)] = mp.to_numpy(np.float32, copy=False)
|
|
192
|
+
i += len(mp)
|
|
193
|
+
del mps
|
|
194
|
+
|
|
195
|
+
allmp -= (allmp.sum(axis=0, dtype=np.float64) / n).astype(np.float32)
|
|
196
|
+
stds = np.sqrt(np.einsum('ij,ij->j', allmp, allmp, dtype=np.float64) / n)
|
|
197
|
+
stds[stds == 0] = 1 # constant features stay exactly 0, as the old fillna(0) left them
|
|
198
|
+
allmp /= stds.astype(np.float32)
|
|
199
|
+
|
|
200
|
+
allmp = ad.AnnData(X=allmp,
|
|
201
|
+
obs=pd.DataFrame(index=np.arange(n).astype(str)),
|
|
202
|
+
var=pd.DataFrame(index=markers))
|
|
156
203
|
logger.info(f'Metapixel matrix: {allmp.shape[0]:,} pixels × {allmp.shape[1]} features')
|
|
157
204
|
|
|
158
205
|
logger.info('performing PCA...')
|
|
@@ -176,7 +223,7 @@ def pca_metapixels(mps, k):
|
|
|
176
223
|
plt.xticks(range(len(loadings.columns)), loadings.columns, rotation=90)
|
|
177
224
|
settings.show('pc_loadings')
|
|
178
225
|
|
|
179
|
-
return loadings,
|
|
226
|
+
return loadings, allmp
|
|
180
227
|
|
|
181
228
|
def pca_pixels(normedpixelsdir, masksdir, pcloadings, sids):
|
|
182
229
|
"""
|
|
@@ -193,16 +240,18 @@ def pca_pixels(normedpixelsdir, masksdir, pcloadings, sids):
|
|
|
193
240
|
column giving the source sample.
|
|
194
241
|
"""
|
|
195
242
|
pcs = []
|
|
196
|
-
|
|
243
|
+
sid_codes = []
|
|
244
|
+
# project in float32: a DataFrame (or float64) right-hand side silently
|
|
245
|
+
# promotes the result, doubling both the projection cost and the size of
|
|
246
|
+
# the returned table, which has a row per pixel in the whole dataset
|
|
247
|
+
loadings = np.ascontiguousarray(np.asarray(pcloadings), dtype=np.float32)
|
|
197
248
|
|
|
198
249
|
logger.info('Applying PCA projection to each sample')
|
|
199
|
-
for sid in settings.progress(sids, name='pixels -> PCA space'):
|
|
250
|
+
for code, sid in enumerate(settings.progress(sids, name='pixels -> PCA space')):
|
|
200
251
|
da = xr.open_dataarray(f'{normedpixelsdir}/{sid}.nc')
|
|
201
252
|
mask_da = xr.open_dataarray(f'{masksdir}/{sid}.nc')
|
|
202
253
|
|
|
203
|
-
means =
|
|
204
|
-
stds = xr.DataArray(da.attrs['stds'], dims='marker')
|
|
205
|
-
da = ((da - means) / stds).where(mask_da, 0)
|
|
254
|
+
means, stds = da.attrs['means'], da.attrs['stds']
|
|
206
255
|
|
|
207
256
|
# load raw arrays and close before dtype conversion so we never hold
|
|
208
257
|
# two full (H × W × n_genes) copies simultaneously
|
|
@@ -213,19 +262,29 @@ def pca_pixels(normedpixelsdir, masksdir, pcloadings, sids):
|
|
|
213
262
|
pl = data.astype(np.float32, copy=False)[mask]
|
|
214
263
|
del data, mask; gc.collect()
|
|
215
264
|
|
|
216
|
-
|
|
265
|
+
# standardize the non-empty pixels only, rather than the full
|
|
266
|
+
# (y, x, marker) array; empty pixels are dropped by the mask anyway
|
|
267
|
+
pl -= means
|
|
268
|
+
pl /= stds
|
|
269
|
+
|
|
270
|
+
pl_pca = pl.dot(loadings)
|
|
217
271
|
pcs.append(pl_pca)
|
|
218
|
-
|
|
272
|
+
sid_codes.append(np.full(pl_pca.shape[0], code, dtype=np.int32))
|
|
219
273
|
del pl; gc.collect()
|
|
220
274
|
|
|
221
275
|
# concatenate
|
|
222
276
|
pcs = np.vstack(pcs)
|
|
223
|
-
|
|
277
|
+
sid_codes = np.concatenate(sid_codes)
|
|
224
278
|
|
|
225
279
|
allpixels_pca = pd.DataFrame(
|
|
226
280
|
pcs,
|
|
227
|
-
columns=[f'PC{i}' for i in range(1,
|
|
281
|
+
columns=[f'PC{i}' for i in range(1, loadings.shape[1] + 1)]
|
|
228
282
|
)
|
|
229
|
-
|
|
283
|
+
# categorical rather than an object column: one code per pixel instead of
|
|
284
|
+
# one pointer, over tens of millions of rows
|
|
285
|
+
# drop categories for samples that contributed no pixels, so downstream
|
|
286
|
+
# get_dummies (Harmony) never sees an all-zero batch column
|
|
287
|
+
allpixels_pca['sid'] = pd.Categorical.from_codes(
|
|
288
|
+
sid_codes, categories=list(sids)).remove_unused_categories()
|
|
230
289
|
|
|
231
290
|
return allpixels_pca
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import numpy as numpy
|
|
2
2
|
import numpy as np
|
|
3
|
+
import pandas as pd
|
|
3
4
|
import xarray as xr
|
|
4
5
|
import cv2 as cv2
|
|
5
6
|
import scanpy as sc
|
|
@@ -105,6 +106,161 @@ def add_covs(pca, sid_to_covs):
|
|
|
105
106
|
pca[cov_name] = pca['sid'].map(sid_to_covs[cov_name])
|
|
106
107
|
return ['sid'] + cov_names
|
|
107
108
|
|
|
109
|
+
def collapse_markers(normeddir, markers, pseudomarker, outdir, masksdir=None):
|
|
110
|
+
"""
|
|
111
|
+
Keep a subset of markers and fold all the others into one pseudomarker.
|
|
112
|
+
|
|
113
|
+
Reads the normalized pixel matrices written by `prepare_merfish`,
|
|
114
|
+
`prepare_xenium5k`, or `nonst.prepare` (for transcriptomic data the markers
|
|
115
|
+
are genes), retains `markers`, and replaces every other marker with a single
|
|
116
|
+
channel named `pseudomarker` holding their combined signal. Because the
|
|
117
|
+
stored values are log-normalized, they are exponentiated, summed, and
|
|
118
|
+
log-normalized again, so the pseudomarker is on the same scale as a real
|
|
119
|
+
marker. The dataset-wide means and stds stored on each file are rewritten to
|
|
120
|
+
match the new marker set: retained markers keep their existing values and the
|
|
121
|
+
pseudomarker's are computed from the collapsed data.
|
|
122
|
+
|
|
123
|
+
Parameters
|
|
124
|
+
----------
|
|
125
|
+
normeddir
|
|
126
|
+
Directory of normalized ``.nc`` files, e.g. ``{outdir}/normalized``.
|
|
127
|
+
markers
|
|
128
|
+
Markers to retain, in the order they should appear; the pseudomarker is
|
|
129
|
+
appended after them. Markers absent from the data are skipped with a
|
|
130
|
+
warning.
|
|
131
|
+
pseudomarker
|
|
132
|
+
Name for the new channel, e.g. ``'nonimmune'``.
|
|
133
|
+
outdir
|
|
134
|
+
Directory to write the collapsed ``.nc`` files to. This is a
|
|
135
|
+
``normalized``-style directory, so to build a dataset that
|
|
136
|
+
`pca_pixels` can be pointed at, pass ``f'{newroot}/normalized'`` and copy
|
|
137
|
+
the masks to ``f'{newroot}/masks'``.
|
|
138
|
+
masksdir
|
|
139
|
+
Directory of tissue masks, used to compute the pseudomarker's moments
|
|
140
|
+
over non-empty pixels only. Defaults to ``masks`` alongside `normeddir`.
|
|
141
|
+
"""
|
|
142
|
+
import netCDF4
|
|
143
|
+
|
|
144
|
+
if masksdir is None:
|
|
145
|
+
masksdir = os.path.join(os.path.dirname(os.path.normpath(normeddir)), 'masks')
|
|
146
|
+
|
|
147
|
+
sids = [os.path.splitext(f)[0]
|
|
148
|
+
for f in os.listdir(normeddir) if f.endswith('.nc') and not f.startswith('.')]
|
|
149
|
+
if len(sids) == 0:
|
|
150
|
+
logger.warning(f'No .nc files found in {normeddir}. Check your path and try again.')
|
|
151
|
+
return
|
|
152
|
+
os.makedirs(outdir, exist_ok=True)
|
|
153
|
+
|
|
154
|
+
# decide on the new marker set, using the first sample as the reference
|
|
155
|
+
da = xr.open_dataarray(f'{normeddir}/{sids[0]}.nc')
|
|
156
|
+
ref_markers = list(da.marker.values)
|
|
157
|
+
ref_means, ref_stds = da.attrs['means'], da.attrs['stds']
|
|
158
|
+
da.close(); del da
|
|
159
|
+
|
|
160
|
+
if pseudomarker in ref_markers:
|
|
161
|
+
raise ValueError(f"'{pseudomarker}' is already a marker; pick a name for the "
|
|
162
|
+
f"pseudomarker that isn't in the data.")
|
|
163
|
+
missing = [m for m in markers if m not in ref_markers]
|
|
164
|
+
if missing:
|
|
165
|
+
logger.warning(f'{len(missing)} of the {len(markers)} requested markers are not in the '
|
|
166
|
+
f'data and will be skipped: {missing}')
|
|
167
|
+
keep = [m for m in markers if m in ref_markers]
|
|
168
|
+
if len(keep) == 0:
|
|
169
|
+
raise ValueError('None of the requested markers are in the data.')
|
|
170
|
+
if len(keep) == len(ref_markers):
|
|
171
|
+
logger.warning(f'All {len(ref_markers)} markers were retained, so {pseudomarker} will be '
|
|
172
|
+
f'empty.')
|
|
173
|
+
logger.info(f'Keeping {len(keep)} markers and collapsing the other '
|
|
174
|
+
f'{len(ref_markers) - len(keep)} into {pseudomarker}.')
|
|
175
|
+
|
|
176
|
+
# collapse each sample, accumulating the pseudomarker's per-sample moments as we go
|
|
177
|
+
sample_sids, sample_means, sample_stds, sample_npixels = [], [], [], []
|
|
178
|
+
for sid in settings.progress(sids, name='collapsing markers'):
|
|
179
|
+
da = xr.open_dataarray(f'{normeddir}/{sid}.nc')
|
|
180
|
+
mask_da = xr.open_dataarray(f'{masksdir}/{sid}.nc')
|
|
181
|
+
|
|
182
|
+
sid_markers = list(da.marker.values)
|
|
183
|
+
if sid_markers != ref_markers:
|
|
184
|
+
logger.warning(f'{sid} has different markers ({len(sid_markers)}) than {sids[0]} '
|
|
185
|
+
f'({len(ref_markers)}); matching by name.')
|
|
186
|
+
absent = [m for m in keep if m not in sid_markers]
|
|
187
|
+
if absent:
|
|
188
|
+
raise ValueError(f'{sid} is missing {len(absent)} of the markers to retain: {absent}')
|
|
189
|
+
ix = {m: i for i, m in enumerate(sid_markers)}
|
|
190
|
+
keep_ix = [ix[m] for m in keep]
|
|
191
|
+
keep_ix_set = set(keep_ix)
|
|
192
|
+
drop_ix = [i for i in range(len(sid_markers)) if i not in keep_ix_set]
|
|
193
|
+
|
|
194
|
+
# load raw arrays and close before doing any work, so we never hold two
|
|
195
|
+
# full (H x W x n_markers) copies simultaneously
|
|
196
|
+
x = da.x.values; y = da.y.values
|
|
197
|
+
data = da.values
|
|
198
|
+
mask = mask_da.values
|
|
199
|
+
da.close(); mask_da.close()
|
|
200
|
+
del da, mask_da; gc.collect()
|
|
201
|
+
|
|
202
|
+
# accumulate the dropped markers one at a time: fancy-indexing them all at
|
|
203
|
+
# once would materialize the (y, x, n_dropped) copy this function exists to
|
|
204
|
+
# avoid. log-normalized values are exponentiated before summing so the
|
|
205
|
+
# pseudomarker ends up on the same scale as the markers we kept.
|
|
206
|
+
acc = np.zeros(data.shape[:2], dtype=np.float64)
|
|
207
|
+
for j in drop_ix:
|
|
208
|
+
acc += np.expm1(data[..., j].astype(np.float64))
|
|
209
|
+
other = np.log1p(acc)
|
|
210
|
+
del acc
|
|
211
|
+
|
|
212
|
+
# empty pixels are zero in the input, and log1p(sum(expm1(0))) is zero, so
|
|
213
|
+
# they stay empty without any special handling
|
|
214
|
+
s = xr.DataArray(
|
|
215
|
+
np.concatenate([data[..., keep_ix], other[..., None].astype(np.float32)], axis=-1),
|
|
216
|
+
dims=['y', 'x', 'marker'],
|
|
217
|
+
coords={'x': x, 'y': y, 'marker': keep + [pseudomarker]})
|
|
218
|
+
s.name = sid
|
|
219
|
+
util.write_xarray(s, f'{outdir}/{sid}.nc')
|
|
220
|
+
del data, s; gc.collect()
|
|
221
|
+
|
|
222
|
+
# a sample with an empty mask contributes nothing rather than a nan that
|
|
223
|
+
# would poison the pooled moments
|
|
224
|
+
vals = other[mask]
|
|
225
|
+
if len(vals) == 0:
|
|
226
|
+
logger.warning(f'{sid} has no non-empty pixels; excluding it from the '
|
|
227
|
+
f'{pseudomarker} moments.')
|
|
228
|
+
else:
|
|
229
|
+
sample_sids.append(sid)
|
|
230
|
+
sample_means.append(vals.mean(dtype=np.float64))
|
|
231
|
+
sample_stds.append(vals.std(dtype=np.float64))
|
|
232
|
+
sample_npixels.append(len(vals))
|
|
233
|
+
del other, mask, vals; gc.collect()
|
|
234
|
+
|
|
235
|
+
if len(sample_sids) == 0:
|
|
236
|
+
raise ValueError(f'No non-empty pixels in any sample, so {pseudomarker} has no moments. '
|
|
237
|
+
f'Check that {masksdir} holds the masks for {normeddir}.')
|
|
238
|
+
|
|
239
|
+
# pool the pseudomarker's moments across samples the same way get_sumstats does
|
|
240
|
+
pseudo_mean, pseudo_std = util.pool_moments(
|
|
241
|
+
pd.DataFrame([sample_means], index=[pseudomarker], columns=sample_sids),
|
|
242
|
+
pd.DataFrame([sample_stds], index=[pseudomarker], columns=sample_sids),
|
|
243
|
+
sample_npixels)
|
|
244
|
+
pseudo_mean, pseudo_std = pseudo_mean.iloc[0], pseudo_std.iloc[0]
|
|
245
|
+
if pseudo_std == 0:
|
|
246
|
+
# a constant channel stays exactly 0 after standardization rather than
|
|
247
|
+
# dividing by zero downstream
|
|
248
|
+
logger.warning(f'{pseudomarker} has zero variance; setting its std to 1.')
|
|
249
|
+
pseudo_std = 1.
|
|
250
|
+
logger.info(f'{pseudomarker}: mean {pseudo_mean:.3f}, std {pseudo_std:.3f}')
|
|
251
|
+
|
|
252
|
+
# the pseudomarker's moments aren't known until every sample has been read, so
|
|
253
|
+
# the files are written above without them and stamped here; this rewrites
|
|
254
|
+
# metadata only, rather than re-collapsing every sample a second time
|
|
255
|
+
keep_ix_ref = [ref_markers.index(m) for m in keep]
|
|
256
|
+
means = np.append(np.asarray(ref_means)[keep_ix_ref], pseudo_mean).astype(np.float32)
|
|
257
|
+
stds = np.append(np.asarray(ref_stds)[keep_ix_ref], pseudo_std).astype(np.float32)
|
|
258
|
+
for sid in sids:
|
|
259
|
+
with netCDF4.Dataset(f'{outdir}/{sid}.nc', 'a') as ds:
|
|
260
|
+
v = ds.variables[sid]
|
|
261
|
+
v.setncattr('means', means)
|
|
262
|
+
v.setncattr('stds', stds)
|
|
263
|
+
|
|
108
264
|
def pca_pixels(outdir, repname, nmetamarkers=10, npixels_to_plot=50000,
|
|
109
265
|
total_n_metapixels=2_000_000, sid_to_covs=None):
|
|
110
266
|
"""
|
|
@@ -149,7 +305,7 @@ def pca_pixels(outdir, repname, nmetamarkers=10, npixels_to_plot=50000,
|
|
|
149
305
|
total_n_metapixels=total_n_metapixels)
|
|
150
306
|
|
|
151
307
|
# PCA the metapixels
|
|
152
|
-
loadings,
|
|
308
|
+
loadings, allmp = dimreduce.pca_metapixels(metapixels.values(), nmetamarkers)
|
|
153
309
|
loadings.to_feather(f'{processeddir}/_pcloadings.feather')
|
|
154
310
|
del metapixels, allmp; gc.collect()
|
|
155
311
|
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import os, glob, gc
|
|
2
2
|
import numpy as np
|
|
3
|
+
import pandas as pd
|
|
3
4
|
import cv2 as cv2
|
|
4
5
|
import xarray as xr
|
|
5
6
|
from . import util
|
|
@@ -134,12 +135,14 @@ def prepare(load, filepaths, orig_pixel_size, markers, get_foreground, norm_by_b
|
|
|
134
135
|
)
|
|
135
136
|
for sid in settings.progress(sids)])
|
|
136
137
|
gc.collect()
|
|
137
|
-
|
|
138
|
+
goodmarkers, pixels = norm_by_background(pixels)
|
|
138
139
|
ntranscripts = pixels.sum(axis=1, dtype=np.float64)
|
|
139
140
|
med_ntranscripts = np.median(ntranscripts)
|
|
140
141
|
pixels = np.log1p(med_ntranscripts * pixels / (ntranscripts[:,None] + 1e-6)) # adding to denominator in case pixel is all 0s
|
|
141
|
-
|
|
142
|
-
|
|
142
|
+
# indexed by marker name so each sample's moments are aligned by name below,
|
|
143
|
+
# rather than positionally against whatever markers that sample kept
|
|
144
|
+
means = pd.Series(pixels.mean(axis=0, dtype=np.float64), index=goodmarkers)
|
|
145
|
+
stds = pd.Series(pixels.std(axis=0, dtype=np.float64), index=goodmarkers)
|
|
143
146
|
del pixels; gc.collect()
|
|
144
147
|
|
|
145
148
|
logger.info('Normalizing and writing')
|
|
@@ -153,6 +156,7 @@ def prepare(load, filepaths, orig_pixel_size, markers, get_foreground, norm_by_b
|
|
|
153
156
|
pl = np.log1p(med_ntranscripts * pl / (pl.sum(axis=1)[:,None] + 1e-6)) # adding to denominator in case pixel is all 0s
|
|
154
157
|
s = s.sel(marker=goodmarkers)
|
|
155
158
|
util.set_pixels(s, mask, pl)
|
|
156
|
-
|
|
157
|
-
s.attrs['
|
|
159
|
+
# a marker with no moments standardizes to a no-op rather than dividing by zero
|
|
160
|
+
s.attrs['means'] = means.reindex(s.marker.values, fill_value=0).values.astype(np.float32)
|
|
161
|
+
s.attrs['stds'] = stds.reindex(s.marker.values, fill_value=1).values.astype(np.float32)
|
|
158
162
|
util.write_xarray(s, f'{normeddir}/{sid}.nc')
|