thyra 2.0.1__tar.gz → 2.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {thyra-2.0.1 → thyra-2.0.2}/PKG-INFO +1 -1
- {thyra-2.0.1 → thyra-2.0.2}/pyproject.toml +1 -1
- {thyra-2.0.1 → thyra-2.0.2}/thyra/__init__.py +1 -1
- {thyra-2.0.1 → thyra-2.0.2}/thyra/converters/spatialdata/streaming_converter.py +41 -7
- {thyra-2.0.1 → thyra-2.0.2}/LICENSE +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/README.md +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/__main__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/alignment/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/alignment/affine.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/alignment/teaching_points.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/config.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/convert.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/converters/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/converters/spatialdata/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/converters/spatialdata/_chunking.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/converters/spatialdata/base_spatialdata_converter.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/converters/spatialdata/converter.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/converters/spatialdata/spatialdata_2d_converter.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/converters/spatialdata/spatialdata_3d_converter.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/core/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/core/base_converter.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/core/base_extractor.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/core/base_reader.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/core/registry.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/extractors/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/extractors/bruker_extractor.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/extractors/imzml_extractor.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/extractors/waters_extractor.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/ontology/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/ontology/_ims.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/ontology/_ms.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/ontology/_uo.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/ontology/cache.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/types.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/metadata/validator.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/preview.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/base_bruker_reader.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/folder_structure.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/mis_parser.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/rapiflex/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/rapiflex/rapiflex_reader.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/sdk/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/sdk/dll/LICENCE-BRUKER.txt +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/sdk/dll/README.md +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/sdk/dll/timsdata.dll +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/sdk/dll/timsdata.so +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/sdk/dll_manager.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/sdk/platform_detector.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/sdk/sdk_functions.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/timstof_reader.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/utils/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/utils/batch_processor.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/utils/coordinate_cache.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/utils/mass_axis_builder.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/bruker/timstof/utils/memory_manager.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/imzml/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/imzml/imzml_reader.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/waters/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/waters/imaging_grid.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/waters/lib/MLReader.dll +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/waters/lib/MassLynxRaw.dll +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/waters/lib/libMLReader.so +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/waters/lib/libMassLynxRaw.so +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/waters/masslynx_lib.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/readers/waters/waters_reader.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/common_axis.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/constants.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/data_characteristics.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/decision_tree.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/instrument_detectors.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/mass_axis/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/mass_axis/base_generator.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/mass_axis/fticr_generator.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/mass_axis/linear_generator.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/mass_axis/linear_tof_generator.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/mass_axis/orbitrap_generator.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/mass_axis/reflector_tof_generator.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/strategies/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/strategies/base.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/strategies/nearest_neighbor.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/strategies/tic_preserving.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/tic.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/resampling/types.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/tools/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/tools/check_ontology.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/tools/make_example_data.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/utils/__init__.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/utils/bruker_exceptions.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/utils/logging_config.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/utils/windows_paths.py +0 -0
- {thyra-2.0.1 → thyra-2.0.2}/thyra/utils/zarr_atomic_write.py +0 -0
|
@@ -4,7 +4,7 @@ build-backend = "poetry.core.masonry.api"
|
|
|
4
4
|
|
|
5
5
|
[tool.poetry]
|
|
6
6
|
name = "thyra"
|
|
7
|
-
version = "2.0.
|
|
7
|
+
version = "2.0.2"
|
|
8
8
|
description = "A modern Python library for converting Mass Spectrometry Imaging (MSI) data into SpatialData/Zarr format - your portal to spatial omics"
|
|
9
9
|
authors = ["Theodoros Visvikis <t.visvikis@maastrichtuniversity.nl>"]
|
|
10
10
|
maintainers = ["Theodoros Visvikis <t.visvikis@maastrichtuniversity.nl>"]
|
|
@@ -40,6 +40,12 @@ if SPATIALDATA_AVAILABLE:
|
|
|
40
40
|
|
|
41
41
|
logger = logging.getLogger(__name__)
|
|
42
42
|
|
|
43
|
+
# Elements per chunk when building the string index arrays. Bounds the
|
|
44
|
+
# transient cost of formatting to this many entries rather than the whole
|
|
45
|
+
# axis; measured at 17.5 bytes per entry at peak against 88 for a one-shot
|
|
46
|
+
# build, and 32 for an unchunked vectorised one.
|
|
47
|
+
_INDEX_BUILD_CHUNK = 1_000_000
|
|
48
|
+
|
|
43
49
|
|
|
44
50
|
class StreamingSpatialDataConverter(BaseSpatialDataConverter):
|
|
45
51
|
"""Memory-efficient streaming converter for MSI data to SpatialData format.
|
|
@@ -491,11 +497,19 @@ class StreamingSpatialDataConverter(BaseSpatialDataConverter):
|
|
|
491
497
|
indptr_arr.attrs["encoding-type"] = "array"
|
|
492
498
|
indptr_arr.attrs["encoding-version"] = "0.2.0"
|
|
493
499
|
|
|
500
|
+
# Column indices are bounded by n_cols, not by total_nnz, so they need
|
|
501
|
+
# their own dtype decision. Hardcoding int32 here wrapped silently
|
|
502
|
+
# above 2,147,483,647 m/z bins: zarr truncates an oversized write
|
|
503
|
+
# without warning, and the resulting negative indices survive all the
|
|
504
|
+
# way into scipy without an error. This is the same switch point scipy
|
|
505
|
+
# uses, so the matrix rebuilt from these arrays needs no cast.
|
|
506
|
+
indices_dtype = np.int64 if n_cols > np.iinfo(np.int32).max else np.int32
|
|
507
|
+
|
|
494
508
|
chunk_size_zarr = min(total_nnz, 1000000)
|
|
495
509
|
indices_arr = X_group.create_array(
|
|
496
510
|
"indices",
|
|
497
511
|
shape=(total_nnz,),
|
|
498
|
-
dtype=
|
|
512
|
+
dtype=indices_dtype,
|
|
499
513
|
chunks=(chunk_size_zarr,),
|
|
500
514
|
)
|
|
501
515
|
indices_arr.attrs["encoding-type"] = "array"
|
|
@@ -566,7 +580,11 @@ class StreamingSpatialDataConverter(BaseSpatialDataConverter):
|
|
|
566
580
|
pos = write_pos[pixel_idx]
|
|
567
581
|
positions = np.arange(pos, pos + nnz)
|
|
568
582
|
buf_positions.append(positions)
|
|
569
|
-
|
|
583
|
+
# Take the dtype from the destination array rather than
|
|
584
|
+
# re-deriving it, so the buffer and the store cannot drift
|
|
585
|
+
# apart. Zarr accepts an oversized write and truncates it
|
|
586
|
+
# silently, so a mismatch here would be invisible.
|
|
587
|
+
buf_indices.append(mz_indices.astype(indices_arr.dtype))
|
|
570
588
|
buf_data.append(resampled_ints.astype(np.float64))
|
|
571
589
|
write_pos[pixel_idx] += nnz
|
|
572
590
|
buf_size += nnz
|
|
@@ -716,11 +734,15 @@ class StreamingSpatialDataConverter(BaseSpatialDataConverter):
|
|
|
716
734
|
|
|
717
735
|
logger.info(f"Loaded CSR components: {len(data):,} entries")
|
|
718
736
|
|
|
719
|
-
#
|
|
737
|
+
# indices already carries the dtype chosen at write time, keyed on the
|
|
738
|
+
# column count (see _coo_setup_zarr_arrays), so there is nothing to fix
|
|
739
|
+
# up here. Upcasting it was never a safeguard anyway: it widened values
|
|
740
|
+
# that had already been truncated on the way in, and it was keyed on
|
|
741
|
+
# the wrong quantity. indptr is bounded by nnz, hence the check below.
|
|
742
|
+
# copy=False makes the no-op case free rather than a full-size copy.
|
|
720
743
|
if len(data) > np.iinfo(np.int32).max:
|
|
721
744
|
logger.info("Large dataset detected, using 64-bit sparse matrix indices")
|
|
722
|
-
indptr = indptr.astype(np.int64)
|
|
723
|
-
indices = indices.astype(np.int64)
|
|
745
|
+
indptr = indptr.astype(np.int64, copy=False)
|
|
724
746
|
|
|
725
747
|
# Create CSR matrix directly (no COO intermediate)
|
|
726
748
|
sparse_matrix: Union[sparse.csr_matrix, sparse.csc_matrix] = sparse.csr_matrix(
|
|
@@ -1417,7 +1439,9 @@ class StreamingSpatialDataConverter(BaseSpatialDataConverter):
|
|
|
1417
1439
|
|
|
1418
1440
|
y_values = np.repeat(np.arange(n_y, dtype=np.int32), n_x)
|
|
1419
1441
|
x_values = np.tile(np.arange(n_x, dtype=np.int32), n_y)
|
|
1420
|
-
|
|
1442
|
+
# Same reasoning as the var index below, but n_rows is the pixel count
|
|
1443
|
+
# and stays far smaller than n_cols, so one shot needs no chunking.
|
|
1444
|
+
instance_ids = np.arange(n_rows, dtype=np.int64).astype(str_dtype)
|
|
1421
1445
|
spatial_x = x_values.astype(np.float64) * self.pixel_size_um
|
|
1422
1446
|
spatial_y = y_values.astype(np.float64) * self.pixel_size_um
|
|
1423
1447
|
|
|
@@ -1464,7 +1488,17 @@ class StreamingSpatialDataConverter(BaseSpatialDataConverter):
|
|
|
1464
1488
|
mz_values = self._common_mass_axis
|
|
1465
1489
|
if mz_values is None:
|
|
1466
1490
|
raise RuntimeError("Common mass axis not initialized")
|
|
1467
|
-
|
|
1491
|
+
# Built in chunks rather than from a list comprehension. Materialising
|
|
1492
|
+
# n_cols Python str objects first costs about 88 bytes per entry at
|
|
1493
|
+
# peak, against 17.5 for this; at 10 million bins that is 883 MB
|
|
1494
|
+
# versus 175 MB, on a path whose docstring promises roughly 200 MB
|
|
1495
|
+
# regardless of dataset size. The values are identical.
|
|
1496
|
+
mz_index = np.empty(n_cols, dtype=str_dtype)
|
|
1497
|
+
for start in range(0, n_cols, _INDEX_BUILD_CHUNK):
|
|
1498
|
+
stop = min(start + _INDEX_BUILD_CHUNK, n_cols)
|
|
1499
|
+
mz_index[start:stop] = np.strings.add(
|
|
1500
|
+
"mz_", np.arange(start, stop, dtype=np.int64).astype(str_dtype)
|
|
1501
|
+
)
|
|
1468
1502
|
a = var_group.create_array("_index", data=mz_index)
|
|
1469
1503
|
a.attrs["encoding-type"] = "string-array"
|
|
1470
1504
|
a.attrs["encoding-version"] = "0.2.0"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|