vima-spatial 0.2.7__tar.gz → 0.2.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/PKG-INFO +1 -1
  2. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/setup.cfg +1 -1
  3. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/ingest/dimreduce.py +108 -49
  4. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/ingest/ingest.py +157 -1
  5. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/ingest/nonst.py +9 -5
  6. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/ingest/st.py +128 -114
  7. vima_spatial-0.2.9/src/vima/ingest/util.py +237 -0
  8. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima_spatial.egg-info/PKG-INFO +1 -1
  9. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima_spatial.egg-info/SOURCES.txt +3 -1
  10. vima_spatial-0.2.9/tests/test_collapse_markers.py +155 -0
  11. vima_spatial-0.2.9/tests/test_st_rasterize.py +363 -0
  12. vima_spatial-0.2.7/src/vima/ingest/util.py +0 -108
  13. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/README.md +0 -0
  14. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/pyproject.toml +0 -0
  15. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/__init__.py +0 -0
  16. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/_settings.py +0 -0
  17. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/cc.py +0 -0
  18. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/data/__init__.py +0 -0
  19. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/data/download.py +0 -0
  20. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/data/patchcollection.py +0 -0
  21. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/data/samples.py +0 -0
  22. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/fingerprints.py +0 -0
  23. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/ingest/__init__.py +0 -0
  24. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/__init__.py +0 -0
  25. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/resnet_vae.py +0 -0
  26. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/resnetlight_decoder.py +0 -0
  27. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/resnetlight_encoder.py +0 -0
  28. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/simple_vae.py +0 -0
  29. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/models/vae.py +0 -0
  30. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/patchfeatures.py +0 -0
  31. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/train/__init__.py +0 -0
  32. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/train/logging.py +0 -0
  33. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/train/training.py +0 -0
  34. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/__init__.py +0 -0
  35. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/features.py +0 -0
  36. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/patches.py +0 -0
  37. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/patchexamples.py +0 -0
  38. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/spatial.py +0 -0
  39. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima/vis/umaps.py +0 -0
  40. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima_spatial.egg-info/dependency_links.txt +0 -0
  41. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima_spatial.egg-info/requires.txt +0 -0
  42. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/src/vima_spatial.egg-info/top_level.txt +0 -0
  43. {vima_spatial-0.2.7 → vima_spatial-0.2.9}/tests/test_ra_regression.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: vima-spatial
3
- Version: 0.2.7
3
+ Version: 0.2.9
4
4
  Summary: variational inference-based microniche analysis
5
5
  Home-page: https://github.com/yakirr/vima
6
6
  Author: Yakir Reshef
@@ -1,6 +1,6 @@
1
1
  [metadata]
2
2
  name = vima-spatial
3
- version = 0.2.7
3
+ version = 0.2.9
4
4
  author = Yakir Reshef
5
5
  author_email = yreshef@broadinstitute.org
6
6
  description = variational inference-based microniche analysis
@@ -16,10 +16,10 @@ def metapixels_allsamples(normedpixelsdir, masksdir, sids, total_n_metapixels):
16
16
  """
17
17
  Pool metapixels across all samples for a more robust PCA fit.
18
18
 
19
- Loads each sample's normalized pixels, standardizes them with the stored
20
- per-marker means/stds, builds metapixels, and randomly downsamples each
21
- sample to roughly ``total_n_metapixels // len(sids)`` metapixels. Warns if a
22
- sample's markers differ from the first sample's.
19
+ Loads each sample's normalized pixels, builds roughly
20
+ ``total_n_metapixels // len(sids)`` randomly chosen metapixels from it, and
21
+ standardizes them with the stored per-marker means/stds. Warns if a sample's
22
+ markers differ from the first sample's.
23
23
 
24
24
  Parameters
25
25
  ----------
@@ -55,7 +55,7 @@ def metapixels_allsamples(normedpixelsdir, masksdir, sids, total_n_metapixels):
55
55
  for i, sid in enumerate(settings.progress(sids, name='creating metapixels')):
56
56
  da = xr.open_dataarray(f'{normedpixelsdir}/{sid}.nc')
57
57
  mask_da = xr.open_dataarray(f'{masksdir}/{sid}.nc')
58
-
58
+
59
59
  # ensure same markers in same order in all files
60
60
  markers = list(da.marker.values)
61
61
  if ref_markers is None:
@@ -68,18 +68,21 @@ def metapixels_allsamples(normedpixelsdir, masksdir, sids, total_n_metapixels):
68
68
  f'{len(missing)} missing, {len(extra)} extra vs {ref_sid}')
69
69
  logger.warning(f'{sid} has different markers ({len(markers)}) '
70
70
  f'than {ref_sid} ({len(ref_markers)}): {detail}')
71
-
72
- means = xr.DataArray(da.attrs['means'], dims='marker')
73
- stds = xr.DataArray(da.attrs['stds'], dims='marker')
74
- da = ((da - means) / stds).where(mask_da, 0)
75
71
 
76
- all_metapixels[sid], all_npixels[sid] = metapixels(da, mask_da)
72
+ means, stds = da.attrs['means'], da.attrs['stds']
73
+
74
+ # metapixels are built from the un-standardized pixels and standardized
75
+ # afterward: averaging over a window and the affine (x-mean)/std commute,
76
+ # so this is identical to standardizing the full array first but avoids
77
+ # materializing several (y, x, marker)-sized temporaries.
78
+ mp, all_npixels[sid] = metapixels(da, mask_da, n_metapixels=nmp_per_sample)
77
79
  da.close(); mask_da.close()
78
80
  del da, mask_da
79
- if len(all_metapixels[sid]) > nmp_per_sample:
80
- ix = np.random.choice(len(all_metapixels[sid]), nmp_per_sample, replace=False)
81
- all_metapixels[sid] = all_metapixels[sid].iloc[ix]
82
- all_npixels[sid] = all_npixels[sid][ix]
81
+
82
+ mp -= means
83
+ mp /= stds
84
+ all_metapixels[sid] = pd.DataFrame(data=mp, columns=markers)
85
+ del mp
83
86
 
84
87
  # visualize distribution of num non-empty pixels per metapixel in this sample
85
88
  if settings.show_plots():
@@ -95,33 +98,61 @@ def metapixels_allsamples(normedpixelsdir, masksdir, sids, total_n_metapixels):
95
98
 
96
99
  return all_metapixels, all_npixels
97
100
 
98
- def metapixels(s, mask, npixels_thresh=0):
101
+ def metapixels(s, mask, npixels_thresh=0, n_metapixels=None, window=5):
99
102
  """
100
103
  Pool each pixel with its neighbors into a metapixel.
101
104
 
102
- Sums each marker over a 5x5 window centered on every pixel and divides by
103
- the number of non-empty (masked) pixels contributing to that window, so each
105
+ Averages each marker over a ``window``-by-``window`` window centered on a
106
+ pixel, using only the non-empty (masked) pixels in that window, so each
104
107
  metapixel is the average over its non-empty neighbors. Metapixels with at
105
108
  most ``npixels_thresh`` contributing pixels are dropped.
106
109
 
110
+ Parameters
111
+ ----------
112
+ n_metapixels
113
+ If given, uniformly sample at most this many metapixel centers and
114
+ compute only those. Since the caller typically keeps a small random
115
+ subset anyway, this avoids convolving the whole (y, x, marker) array,
116
+ which dominates the cost for large marker panels.
117
+
107
118
  Returns
108
119
  -------
109
120
  tuple
110
- ``(metapixel_df, npixels)``: a marker-columned DataFrame of retained
111
- metapixels and the per-metapixel count of contributing non-empty pixels.
121
+ ``(metapixels, npixels)``: a ``(metapixel, marker)`` float32 array and
122
+ the per-metapixel count of contributing non-empty pixels.
112
123
  """
113
- markers = s.marker.values
114
-
115
- # make metapixels and compute how many non-empty pixels and transcripts are in each metapixel
116
- kernel = np.ones((5, 5), np.float32)
117
- mp = convolve(s.data, kernel[:, :, None], mode="constant")
118
- npixels = convolve(mask.data.astype('float32'), kernel, mode="constant")
119
-
120
- # filter out metapixels with few non-empty pixels
121
- metapixels_mask = npixels > npixels_thresh
124
+ mask = mask.data
125
+ H, W = mask.shape
126
+
127
+ # how many non-empty pixels contribute to each candidate metapixel (cheap: 2D only)
128
+ kernel = np.ones((window, window), np.float32)
129
+ npixels = convolve(mask.astype(np.float32), kernel, mode="constant")
130
+
131
+ # pick the metapixel centers, sampling before doing any work over markers
132
+ centers = np.flatnonzero(npixels.ravel() > npixels_thresh)
133
+ if n_metapixels is not None and len(centers) > n_metapixels:
134
+ centers = centers[np.random.choice(len(centers), n_metapixels, replace=False)]
135
+ npixels = npixels.ravel()[centers]
136
+
137
+ # sum each window by gathering its non-empty pixels, one neighbor offset at a time
138
+ data = s.data.reshape(H * W, -1)
139
+ mask = mask.ravel()
140
+ r, c = np.divmod(centers, W)
141
+ mp = np.zeros((len(centers), data.shape[1]), np.float32)
142
+ rad = window // 2
143
+ for dr in range(-rad, rad + 1):
144
+ rr = r + dr
145
+ for dc in range(-rad, rad + 1):
146
+ cc = c + dc
147
+ neighbor = rr * W + cc
148
+ contributes = (rr >= 0) & (rr < H) & (cc >= 0) & (cc < W)
149
+ contributes &= mask[np.where(contributes, neighbor, 0)]
150
+ i = np.flatnonzero(contributes)
151
+ mp[i] += data[neighbor[i]]
122
152
 
123
153
  # divide each metapixel by the # of non-empty pixels that contributed to it and return
124
- return pd.DataFrame(data=mp[metapixels_mask] / npixels[metapixels_mask][:,None], columns=markers), npixels[metapixels_mask]
154
+ mp /= npixels[:, None]
155
+ return mp, npixels
125
156
 
126
157
  # mps should be an array of dataframes containing metapixels
127
158
  def pca_metapixels(mps, k):
@@ -142,17 +173,33 @@ def pca_metapixels(mps, k):
142
173
  Returns
143
174
  -------
144
175
  tuple
145
- ``(loadings, C, allmp)``: the gene-by-component loading matrix, the
146
- feature correlation matrix, and the standardized metapixel AnnData.
176
+ ``(loadings, allmp)``: the gene-by-component loading matrix and the
177
+ standardized metapixel AnnData.
147
178
  """
148
179
  logger.info('merging and standardizing metapixels')
149
- allmp = pd.concat(mps)
150
- allmp -= allmp.values.mean(axis=0, dtype=np.float64)
151
- allmp /= allmp.values.std(axis=0, dtype=np.float64)
152
- allmp = allmp.fillna(0)
153
- allmp.index = np.arange(len(allmp)).astype(str)
154
- allmp = ad.AnnData(X=allmp)
155
- C = np.corrcoef(allmp.X[::max(1,(len(allmp)//50000))].T)
180
+ mps = list(mps)
181
+ markers = mps[0].columns
182
+ n = sum(len(mp) for mp in mps)
183
+
184
+ # merge into one preallocated float32 matrix and standardize it in place,
185
+ # accumulating the moments in float64. Doing this in pandas instead upcasts
186
+ # the whole matrix to float64 and copies it once per operation, which at
187
+ # these sizes costs more than the PCA itself.
188
+ allmp = np.empty((n, len(markers)), np.float32)
189
+ i = 0
190
+ for mp in mps:
191
+ allmp[i:i+len(mp)] = mp.to_numpy(np.float32, copy=False)
192
+ i += len(mp)
193
+ del mps
194
+
195
+ allmp -= (allmp.sum(axis=0, dtype=np.float64) / n).astype(np.float32)
196
+ stds = np.sqrt(np.einsum('ij,ij->j', allmp, allmp, dtype=np.float64) / n)
197
+ stds[stds == 0] = 1 # constant features stay exactly 0, as the old fillna(0) left them
198
+ allmp /= stds.astype(np.float32)
199
+
200
+ allmp = ad.AnnData(X=allmp,
201
+ obs=pd.DataFrame(index=np.arange(n).astype(str)),
202
+ var=pd.DataFrame(index=markers))
156
203
  logger.info(f'Metapixel matrix: {allmp.shape[0]:,} pixels × {allmp.shape[1]} features')
157
204
 
158
205
  logger.info('performing PCA...')
@@ -176,7 +223,7 @@ def pca_metapixels(mps, k):
176
223
  plt.xticks(range(len(loadings.columns)), loadings.columns, rotation=90)
177
224
  settings.show('pc_loadings')
178
225
 
179
- return loadings, C, allmp
226
+ return loadings, allmp
180
227
 
181
228
  def pca_pixels(normedpixelsdir, masksdir, pcloadings, sids):
182
229
  """
@@ -193,16 +240,18 @@ def pca_pixels(normedpixelsdir, masksdir, pcloadings, sids):
193
240
  column giving the source sample.
194
241
  """
195
242
  pcs = []
196
- sid_labels = []
243
+ sid_codes = []
244
+ # project in float32: a DataFrame (or float64) right-hand side silently
245
+ # promotes the result, doubling both the projection cost and the size of
246
+ # the returned table, which has a row per pixel in the whole dataset
247
+ loadings = np.ascontiguousarray(np.asarray(pcloadings), dtype=np.float32)
197
248
 
198
249
  logger.info('Applying PCA projection to each sample')
199
- for sid in settings.progress(sids, name='pixels -> PCA space'):
250
+ for code, sid in enumerate(settings.progress(sids, name='pixels -> PCA space')):
200
251
  da = xr.open_dataarray(f'{normedpixelsdir}/{sid}.nc')
201
252
  mask_da = xr.open_dataarray(f'{masksdir}/{sid}.nc')
202
253
 
203
- means = xr.DataArray(da.attrs['means'], dims='marker')
204
- stds = xr.DataArray(da.attrs['stds'], dims='marker')
205
- da = ((da - means) / stds).where(mask_da, 0)
254
+ means, stds = da.attrs['means'], da.attrs['stds']
206
255
 
207
256
  # load raw arrays and close before dtype conversion so we never hold
208
257
  # two full (H × W × n_genes) copies simultaneously
@@ -213,19 +262,29 @@ def pca_pixels(normedpixelsdir, masksdir, pcloadings, sids):
213
262
  pl = data.astype(np.float32, copy=False)[mask]
214
263
  del data, mask; gc.collect()
215
264
 
216
- pl_pca = pl.dot(pcloadings)
265
+ # standardize the non-empty pixels only, rather than the full
266
+ # (y, x, marker) array; empty pixels are dropped by the mask anyway
267
+ pl -= means
268
+ pl /= stds
269
+
270
+ pl_pca = pl.dot(loadings)
217
271
  pcs.append(pl_pca)
218
- sid_labels.append(np.full(pl_pca.shape[0], sid, dtype=object))
272
+ sid_codes.append(np.full(pl_pca.shape[0], code, dtype=np.int32))
219
273
  del pl; gc.collect()
220
274
 
221
275
  # concatenate
222
276
  pcs = np.vstack(pcs)
223
- sid_labels = np.concatenate(sid_labels)
277
+ sid_codes = np.concatenate(sid_codes)
224
278
 
225
279
  allpixels_pca = pd.DataFrame(
226
280
  pcs,
227
- columns=[f'PC{i}' for i in range(1, pcloadings.shape[1] + 1)]
281
+ columns=[f'PC{i}' for i in range(1, loadings.shape[1] + 1)]
228
282
  )
229
- allpixels_pca['sid'] = sid_labels
283
+ # categorical rather than an object column: one code per pixel instead of
284
+ # one pointer, over tens of millions of rows
285
+ # drop categories for samples that contributed no pixels, so downstream
286
+ # get_dummies (Harmony) never sees an all-zero batch column
287
+ allpixels_pca['sid'] = pd.Categorical.from_codes(
288
+ sid_codes, categories=list(sids)).remove_unused_categories()
230
289
 
231
290
  return allpixels_pca
@@ -1,5 +1,6 @@
1
1
  import numpy as numpy
2
2
  import numpy as np
3
+ import pandas as pd
3
4
  import xarray as xr
4
5
  import cv2 as cv2
5
6
  import scanpy as sc
@@ -105,6 +106,161 @@ def add_covs(pca, sid_to_covs):
105
106
  pca[cov_name] = pca['sid'].map(sid_to_covs[cov_name])
106
107
  return ['sid'] + cov_names
107
108
 
109
+ def collapse_markers(normeddir, markers, pseudomarker, outdir, masksdir=None):
110
+ """
111
+ Keep a subset of markers and fold all the others into one pseudomarker.
112
+
113
+ Reads the normalized pixel matrices written by `prepare_merfish`,
114
+ `prepare_xenium5k`, or `nonst.prepare` (for transcriptomic data the markers
115
+ are genes), retains `markers`, and replaces every other marker with a single
116
+ channel named `pseudomarker` holding their combined signal. Because the
117
+ stored values are log-normalized, they are exponentiated, summed, and
118
+ log-normalized again, so the pseudomarker is on the same scale as a real
119
+ marker. The dataset-wide means and stds stored on each file are rewritten to
120
+ match the new marker set: retained markers keep their existing values and the
121
+ pseudomarker's are computed from the collapsed data.
122
+
123
+ Parameters
124
+ ----------
125
+ normeddir
126
+ Directory of normalized ``.nc`` files, e.g. ``{outdir}/normalized``.
127
+ markers
128
+ Markers to retain, in the order they should appear; the pseudomarker is
129
+ appended after them. Markers absent from the data are skipped with a
130
+ warning.
131
+ pseudomarker
132
+ Name for the new channel, e.g. ``'nonimmune'``.
133
+ outdir
134
+ Directory to write the collapsed ``.nc`` files to. This is a
135
+ ``normalized``-style directory, so to build a dataset that
136
+ `pca_pixels` can be pointed at, pass ``f'{newroot}/normalized'`` and copy
137
+ the masks to ``f'{newroot}/masks'``.
138
+ masksdir
139
+ Directory of tissue masks, used to compute the pseudomarker's moments
140
+ over non-empty pixels only. Defaults to ``masks`` alongside `normeddir`.
141
+ """
142
+ import netCDF4
143
+
144
+ if masksdir is None:
145
+ masksdir = os.path.join(os.path.dirname(os.path.normpath(normeddir)), 'masks')
146
+
147
+ sids = [os.path.splitext(f)[0]
148
+ for f in os.listdir(normeddir) if f.endswith('.nc') and not f.startswith('.')]
149
+ if len(sids) == 0:
150
+ logger.warning(f'No .nc files found in {normeddir}. Check your path and try again.')
151
+ return
152
+ os.makedirs(outdir, exist_ok=True)
153
+
154
+ # decide on the new marker set, using the first sample as the reference
155
+ da = xr.open_dataarray(f'{normeddir}/{sids[0]}.nc')
156
+ ref_markers = list(da.marker.values)
157
+ ref_means, ref_stds = da.attrs['means'], da.attrs['stds']
158
+ da.close(); del da
159
+
160
+ if pseudomarker in ref_markers:
161
+ raise ValueError(f"'{pseudomarker}' is already a marker; pick a name for the "
162
+ f"pseudomarker that isn't in the data.")
163
+ missing = [m for m in markers if m not in ref_markers]
164
+ if missing:
165
+ logger.warning(f'{len(missing)} of the {len(markers)} requested markers are not in the '
166
+ f'data and will be skipped: {missing}')
167
+ keep = [m for m in markers if m in ref_markers]
168
+ if len(keep) == 0:
169
+ raise ValueError('None of the requested markers are in the data.')
170
+ if len(keep) == len(ref_markers):
171
+ logger.warning(f'All {len(ref_markers)} markers were retained, so {pseudomarker} will be '
172
+ f'empty.')
173
+ logger.info(f'Keeping {len(keep)} markers and collapsing the other '
174
+ f'{len(ref_markers) - len(keep)} into {pseudomarker}.')
175
+
176
+ # collapse each sample, accumulating the pseudomarker's per-sample moments as we go
177
+ sample_sids, sample_means, sample_stds, sample_npixels = [], [], [], []
178
+ for sid in settings.progress(sids, name='collapsing markers'):
179
+ da = xr.open_dataarray(f'{normeddir}/{sid}.nc')
180
+ mask_da = xr.open_dataarray(f'{masksdir}/{sid}.nc')
181
+
182
+ sid_markers = list(da.marker.values)
183
+ if sid_markers != ref_markers:
184
+ logger.warning(f'{sid} has different markers ({len(sid_markers)}) than {sids[0]} '
185
+ f'({len(ref_markers)}); matching by name.')
186
+ absent = [m for m in keep if m not in sid_markers]
187
+ if absent:
188
+ raise ValueError(f'{sid} is missing {len(absent)} of the markers to retain: {absent}')
189
+ ix = {m: i for i, m in enumerate(sid_markers)}
190
+ keep_ix = [ix[m] for m in keep]
191
+ keep_ix_set = set(keep_ix)
192
+ drop_ix = [i for i in range(len(sid_markers)) if i not in keep_ix_set]
193
+
194
+ # load raw arrays and close before doing any work, so we never hold two
195
+ # full (H x W x n_markers) copies simultaneously
196
+ x = da.x.values; y = da.y.values
197
+ data = da.values
198
+ mask = mask_da.values
199
+ da.close(); mask_da.close()
200
+ del da, mask_da; gc.collect()
201
+
202
+ # accumulate the dropped markers one at a time: fancy-indexing them all at
203
+ # once would materialize the (y, x, n_dropped) copy this function exists to
204
+ # avoid. log-normalized values are exponentiated before summing so the
205
+ # pseudomarker ends up on the same scale as the markers we kept.
206
+ acc = np.zeros(data.shape[:2], dtype=np.float64)
207
+ for j in drop_ix:
208
+ acc += np.expm1(data[..., j].astype(np.float64))
209
+ other = np.log1p(acc)
210
+ del acc
211
+
212
+ # empty pixels are zero in the input, and log1p(sum(expm1(0))) is zero, so
213
+ # they stay empty without any special handling
214
+ s = xr.DataArray(
215
+ np.concatenate([data[..., keep_ix], other[..., None].astype(np.float32)], axis=-1),
216
+ dims=['y', 'x', 'marker'],
217
+ coords={'x': x, 'y': y, 'marker': keep + [pseudomarker]})
218
+ s.name = sid
219
+ util.write_xarray(s, f'{outdir}/{sid}.nc')
220
+ del data, s; gc.collect()
221
+
222
+ # a sample with an empty mask contributes nothing rather than a nan that
223
+ # would poison the pooled moments
224
+ vals = other[mask]
225
+ if len(vals) == 0:
226
+ logger.warning(f'{sid} has no non-empty pixels; excluding it from the '
227
+ f'{pseudomarker} moments.')
228
+ else:
229
+ sample_sids.append(sid)
230
+ sample_means.append(vals.mean(dtype=np.float64))
231
+ sample_stds.append(vals.std(dtype=np.float64))
232
+ sample_npixels.append(len(vals))
233
+ del other, mask, vals; gc.collect()
234
+
235
+ if len(sample_sids) == 0:
236
+ raise ValueError(f'No non-empty pixels in any sample, so {pseudomarker} has no moments. '
237
+ f'Check that {masksdir} holds the masks for {normeddir}.')
238
+
239
+ # pool the pseudomarker's moments across samples the same way get_sumstats does
240
+ pseudo_mean, pseudo_std = util.pool_moments(
241
+ pd.DataFrame([sample_means], index=[pseudomarker], columns=sample_sids),
242
+ pd.DataFrame([sample_stds], index=[pseudomarker], columns=sample_sids),
243
+ sample_npixels)
244
+ pseudo_mean, pseudo_std = pseudo_mean.iloc[0], pseudo_std.iloc[0]
245
+ if pseudo_std == 0:
246
+ # a constant channel stays exactly 0 after standardization rather than
247
+ # dividing by zero downstream
248
+ logger.warning(f'{pseudomarker} has zero variance; setting its std to 1.')
249
+ pseudo_std = 1.
250
+ logger.info(f'{pseudomarker}: mean {pseudo_mean:.3f}, std {pseudo_std:.3f}')
251
+
252
+ # the pseudomarker's moments aren't known until every sample has been read, so
253
+ # the files are written above without them and stamped here; this rewrites
254
+ # metadata only, rather than re-collapsing every sample a second time
255
+ keep_ix_ref = [ref_markers.index(m) for m in keep]
256
+ means = np.append(np.asarray(ref_means)[keep_ix_ref], pseudo_mean).astype(np.float32)
257
+ stds = np.append(np.asarray(ref_stds)[keep_ix_ref], pseudo_std).astype(np.float32)
258
+ for sid in sids:
259
+ with netCDF4.Dataset(f'{outdir}/{sid}.nc', 'a') as ds:
260
+ v = ds.variables[sid]
261
+ v.setncattr('means', means)
262
+ v.setncattr('stds', stds)
263
+
108
264
  def pca_pixels(outdir, repname, nmetamarkers=10, npixels_to_plot=50000,
109
265
  total_n_metapixels=2_000_000, sid_to_covs=None):
110
266
  """
@@ -149,7 +305,7 @@ def pca_pixels(outdir, repname, nmetamarkers=10, npixels_to_plot=50000,
149
305
  total_n_metapixels=total_n_metapixels)
150
306
 
151
307
  # PCA the metapixels
152
- loadings, C, allmp = dimreduce.pca_metapixels(metapixels.values(), nmetamarkers)
308
+ loadings, allmp = dimreduce.pca_metapixels(metapixels.values(), nmetamarkers)
153
309
  loadings.to_feather(f'{processeddir}/_pcloadings.feather')
154
310
  del metapixels, allmp; gc.collect()
155
311
 
@@ -1,5 +1,6 @@
1
1
  import os, glob, gc
2
2
  import numpy as np
3
+ import pandas as pd
3
4
  import cv2 as cv2
4
5
  import xarray as xr
5
6
  from . import util
@@ -134,12 +135,14 @@ def prepare(load, filepaths, orig_pixel_size, markers, get_foreground, norm_by_b
134
135
  )
135
136
  for sid in settings.progress(sids)])
136
137
  gc.collect()
137
- _, pixels = norm_by_background(pixels)
138
+ goodmarkers, pixels = norm_by_background(pixels)
138
139
  ntranscripts = pixels.sum(axis=1, dtype=np.float64)
139
140
  med_ntranscripts = np.median(ntranscripts)
140
141
  pixels = np.log1p(med_ntranscripts * pixels / (ntranscripts[:,None] + 1e-6)) # adding to denominator in case pixel is all 0s
141
- means = pixels.mean(axis=0, dtype=np.float64)
142
- stds = pixels.std(axis=0, dtype=np.float64)
142
+ # indexed by marker name so each sample's moments are aligned by name below,
143
+ # rather than positionally against whatever markers that sample kept
144
+ means = pd.Series(pixels.mean(axis=0, dtype=np.float64), index=goodmarkers)
145
+ stds = pd.Series(pixels.std(axis=0, dtype=np.float64), index=goodmarkers)
143
146
  del pixels; gc.collect()
144
147
 
145
148
  logger.info('Normalizing and writing')
@@ -153,6 +156,7 @@ def prepare(load, filepaths, orig_pixel_size, markers, get_foreground, norm_by_b
153
156
  pl = np.log1p(med_ntranscripts * pl / (pl.sum(axis=1)[:,None] + 1e-6)) # adding to denominator in case pixel is all 0s
154
157
  s = s.sel(marker=goodmarkers)
155
158
  util.set_pixels(s, mask, pl)
156
- s.attrs['means'] = means.reindex(s.marker.values, fill_value=1).values.astype(np.float32)
157
- s.attrs['stds'] = stds.reindex(s.marker.values, fill_value=0).values.astype(np.float32)
159
+ # a marker with no moments standardizes to a no-op rather than dividing by zero
160
+ s.attrs['means'] = means.reindex(s.marker.values, fill_value=0).values.astype(np.float32)
161
+ s.attrs['stds'] = stds.reindex(s.marker.values, fill_value=1).values.astype(np.float32)
158
162
  util.write_xarray(s, f'{normeddir}/{sid}.nc')