thyra 2.3.0__tar.gz → 2.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. {thyra-2.3.0 → thyra-2.3.1}/PKG-INFO +1 -1
  2. {thyra-2.3.0 → thyra-2.3.1}/pyproject.toml +1 -1
  3. {thyra-2.3.0 → thyra-2.3.1}/thyra/__init__.py +1 -1
  4. {thyra-2.3.0 → thyra-2.3.1}/thyra/convert.py +33 -9
  5. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/imzml/imzml_reader.py +486 -6
  6. {thyra-2.3.0 → thyra-2.3.1}/LICENSE +0 -0
  7. {thyra-2.3.0 → thyra-2.3.1}/README.md +0 -0
  8. {thyra-2.3.0 → thyra-2.3.1}/thyra/__main__.py +0 -0
  9. {thyra-2.3.0 → thyra-2.3.1}/thyra/alignment/__init__.py +0 -0
  10. {thyra-2.3.0 → thyra-2.3.1}/thyra/alignment/affine.py +0 -0
  11. {thyra-2.3.0 → thyra-2.3.1}/thyra/alignment/teaching_points.py +0 -0
  12. {thyra-2.3.0 → thyra-2.3.1}/thyra/config.py +0 -0
  13. {thyra-2.3.0 → thyra-2.3.1}/thyra/converters/__init__.py +0 -0
  14. {thyra-2.3.0 → thyra-2.3.1}/thyra/converters/spatialdata/__init__.py +0 -0
  15. {thyra-2.3.0 → thyra-2.3.1}/thyra/converters/spatialdata/_chunking.py +0 -0
  16. {thyra-2.3.0 → thyra-2.3.1}/thyra/converters/spatialdata/base_spatialdata_converter.py +0 -0
  17. {thyra-2.3.0 → thyra-2.3.1}/thyra/converters/spatialdata/converter.py +0 -0
  18. {thyra-2.3.0 → thyra-2.3.1}/thyra/converters/spatialdata/spatialdata_2d_converter.py +0 -0
  19. {thyra-2.3.0 → thyra-2.3.1}/thyra/converters/spatialdata/spatialdata_3d_converter.py +0 -0
  20. {thyra-2.3.0 → thyra-2.3.1}/thyra/converters/spatialdata/streaming_converter.py +0 -0
  21. {thyra-2.3.0 → thyra-2.3.1}/thyra/core/__init__.py +0 -0
  22. {thyra-2.3.0 → thyra-2.3.1}/thyra/core/base_converter.py +0 -0
  23. {thyra-2.3.0 → thyra-2.3.1}/thyra/core/base_extractor.py +0 -0
  24. {thyra-2.3.0 → thyra-2.3.1}/thyra/core/base_reader.py +0 -0
  25. {thyra-2.3.0 → thyra-2.3.1}/thyra/core/registry.py +0 -0
  26. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/__init__.py +0 -0
  27. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/extractors/__init__.py +0 -0
  28. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/extractors/bruker_extractor.py +0 -0
  29. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/extractors/imzml_extractor.py +0 -0
  30. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/extractors/waters_extractor.py +0 -0
  31. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/ontology/__init__.py +0 -0
  32. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/ontology/_ims.py +0 -0
  33. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/ontology/_ms.py +0 -0
  34. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/ontology/_uo.py +0 -0
  35. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/ontology/cache.py +0 -0
  36. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/types.py +0 -0
  37. {thyra-2.3.0 → thyra-2.3.1}/thyra/metadata/validator.py +0 -0
  38. {thyra-2.3.0 → thyra-2.3.1}/thyra/preview.py +0 -0
  39. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/__init__.py +0 -0
  40. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/__init__.py +0 -0
  41. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/base_bruker_reader.py +0 -0
  42. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/folder_structure.py +0 -0
  43. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/mis_parser.py +0 -0
  44. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/rapiflex/__init__.py +0 -0
  45. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/rapiflex/rapiflex_reader.py +0 -0
  46. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/__init__.py +0 -0
  47. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/sdk/__init__.py +0 -0
  48. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/sdk/dll/LICENCE-BRUKER.txt +0 -0
  49. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/sdk/dll/README.md +0 -0
  50. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/sdk/dll/timsdata.dll +0 -0
  51. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/sdk/dll/timsdata.so +0 -0
  52. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/sdk/dll_manager.py +0 -0
  53. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/sdk/platform_detector.py +0 -0
  54. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/sdk/sdk_functions.py +0 -0
  55. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/timstof_reader.py +0 -0
  56. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/utils/__init__.py +0 -0
  57. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/utils/batch_processor.py +0 -0
  58. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/utils/coordinate_cache.py +0 -0
  59. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/utils/mass_axis_builder.py +0 -0
  60. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/bruker/timstof/utils/memory_manager.py +0 -0
  61. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/imzml/__init__.py +0 -0
  62. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/waters/__init__.py +0 -0
  63. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/waters/imaging_grid.py +0 -0
  64. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/waters/lib/MLReader.dll +0 -0
  65. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/waters/lib/MassLynxRaw.dll +0 -0
  66. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/waters/lib/libMLReader.so +0 -0
  67. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/waters/lib/libMassLynxRaw.so +0 -0
  68. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/waters/masslynx_lib.py +0 -0
  69. {thyra-2.3.0 → thyra-2.3.1}/thyra/readers/waters/waters_reader.py +0 -0
  70. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/__init__.py +0 -0
  71. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/common_axis.py +0 -0
  72. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/constants.py +0 -0
  73. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/data_characteristics.py +0 -0
  74. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/decision_tree.py +0 -0
  75. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/gaps.py +0 -0
  76. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/instrument_detectors.py +0 -0
  77. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/mass_axis/__init__.py +0 -0
  78. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/mass_axis/base_generator.py +0 -0
  79. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/mass_axis/fticr_generator.py +0 -0
  80. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/mass_axis/linear_generator.py +0 -0
  81. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/mass_axis/linear_tof_generator.py +0 -0
  82. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/mass_axis/orbitrap_generator.py +0 -0
  83. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/mass_axis/reflector_tof_generator.py +0 -0
  84. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/strategies/__init__.py +0 -0
  85. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/strategies/base.py +0 -0
  86. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/strategies/nearest_neighbor.py +0 -0
  87. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/strategies/tic_preserving.py +0 -0
  88. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/tic.py +0 -0
  89. {thyra-2.3.0 → thyra-2.3.1}/thyra/resampling/types.py +0 -0
  90. {thyra-2.3.0 → thyra-2.3.1}/thyra/tools/__init__.py +0 -0
  91. {thyra-2.3.0 → thyra-2.3.1}/thyra/tools/check_ontology.py +0 -0
  92. {thyra-2.3.0 → thyra-2.3.1}/thyra/tools/make_example_data.py +0 -0
  93. {thyra-2.3.0 → thyra-2.3.1}/thyra/utils/__init__.py +0 -0
  94. {thyra-2.3.0 → thyra-2.3.1}/thyra/utils/bruker_exceptions.py +0 -0
  95. {thyra-2.3.0 → thyra-2.3.1}/thyra/utils/logging_config.py +0 -0
  96. {thyra-2.3.0 → thyra-2.3.1}/thyra/utils/windows_paths.py +0 -0
  97. {thyra-2.3.0 → thyra-2.3.1}/thyra/utils/zarr_atomic_write.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: thyra
3
- Version: 2.3.0
3
+ Version: 2.3.1
4
4
  Summary: A modern Python library for converting Mass Spectrometry Imaging (MSI) data into SpatialData/Zarr format - your portal to spatial omics
5
5
  License: MIT
6
6
  License-File: LICENSE
@@ -4,7 +4,7 @@ build-backend = "poetry.core.masonry.api"
4
4
 
5
5
  [tool.poetry]
6
6
  name = "thyra"
7
- version = "2.3.0"
7
+ version = "2.3.1"
8
8
  description = "A modern Python library for converting Mass Spectrometry Imaging (MSI) data into SpatialData/Zarr format - your portal to spatial omics"
9
9
  authors = ["Theodoros Visvikis <t.visvikis@maastrichtuniversity.nl>"]
10
10
  maintainers = ["Theodoros Visvikis <t.visvikis@maastrichtuniversity.nl>"]
@@ -36,7 +36,7 @@ warnings.filterwarnings(
36
36
  category=FutureWarning,
37
37
  )
38
38
 
39
- __version__ = "2.3.0"
39
+ __version__ = "2.3.1"
40
40
 
41
41
  # Import key components - avoid wildcard imports
42
42
  try:
@@ -147,27 +147,51 @@ def _determine_pixel_size(
147
147
 
148
148
 
149
149
  def _should_use_streaming(streaming: Union[bool, Literal["auto"]], reader: Any) -> bool:
150
- """Determine if streaming converter should be used."""
150
+ """Determine if streaming converter should be used.
151
+
152
+ Args:
153
+ streaming: True to force streaming, False to force the standard
154
+ converter, ``"auto"`` to pick on estimated size.
155
+ reader: The reader for the input.
156
+
157
+ Returns:
158
+ True if the streaming converter should be used.
159
+
160
+ Raises:
161
+ Exception: Whatever ``reader.get_essential_metadata()`` raises. A
162
+ reader that refuses its input must not be silenced here.
163
+ """
151
164
  if streaming is True:
152
165
  return True
153
166
  if streaming != "auto":
154
167
  return False
155
168
 
156
- # Auto-detect based on estimated dataset size (>10GB)
169
+ # Deliberately outside the try below. This is often the first call that
170
+ # builds the reader's parser, so it is the call a refused file fails on --
171
+ # and swallowing it here logged the reason at DEBUG, returned False, and
172
+ # let the converter parse the whole file a second time before failing the
173
+ # same way. On a 2.1 GB imzML that is about two minutes of apparent
174
+ # progress with the real reason invisible.
175
+ essential_meta = reader.get_essential_metadata()
176
+
177
+ # Auto-detect based on estimated dataset size (>10GB). Only the estimate
178
+ # itself is best-effort: a reader whose dimensions are missing or oddly
179
+ # shaped simply does not get the automatic upgrade.
157
180
  try:
158
- essential_meta = reader.get_essential_metadata()
159
181
  dims = essential_meta.dimensions
160
182
  n_pixels = dims[0] * dims[1] * dims[2]
161
183
  # Rough estimate: assume average 10k peaks per spectrum, 8 bytes each
162
184
  estimated_gb = (n_pixels * 10000 * 8) / (1024**3)
163
- if estimated_gb > 10:
164
- logger.info(
165
- f"Auto-detected large dataset (~{estimated_gb:.1f} GB), "
166
- "using streaming converter"
167
- )
168
- return True
169
185
  except Exception as e:
170
186
  logger.debug(f"Could not estimate dataset size for auto-streaming: {e}")
187
+ return False
188
+
189
+ if estimated_gb > 10:
190
+ logger.info(
191
+ f"Auto-detected large dataset (~{estimated_gb:.1f} GB), "
192
+ "using streaming converter"
193
+ )
194
+ return True
171
195
  return False
172
196
 
173
197
 
@@ -1,7 +1,13 @@
1
1
  # thyra/readers/imzml/imzml_reader.py
2
2
  import logging
3
3
  from pathlib import Path
4
- from typing import Any, Dict, Generator, List, Optional, Tuple, Union, cast
4
+ from typing import Any, Dict, Generator, List, NamedTuple, Optional, Tuple, Union, cast
5
+
6
+ # The stdlib XML parser, used to re-read one element of a document pyimzml has
7
+ # already parsed with the same stdlib parser -- see
8
+ # _first_spectrum_array_lengths. Thyra has no defusedxml dependency, and
9
+ # adding one here would not change what has already been read.
10
+ from xml.etree import ElementTree # nosec B405
5
11
 
6
12
  import numpy as np
7
13
  from numpy.typing import NDArray
@@ -38,6 +44,138 @@ _MASS_AXIS_PROBE_BLOCK = 1 << 20
38
44
  # pass None for the old unlimited behaviour.
39
45
  DEFAULT_MAX_MASS_AXIS_LENGTH = 10_000_000
40
46
 
47
+ # mzML's XML namespace, spelled the way pyimzml spells it.
48
+ _MZML_NS = "{http://psi.hupo.org/ms/mzml}"
49
+
50
+ # Number of values in a binary array, and the number of bytes those values
51
+ # occupy once encoded. pyimzml keeps the former and discards the latter.
52
+ _ARRAY_LENGTH_ACCESSION = "IMS:1000103"
53
+ _ENCODED_LENGTH_ACCESSION = "IMS:1000104"
54
+
55
+ # zlib-compressed binary arrays. The strings "zlib", "decompress", "1000574"
56
+ # and "1000104" appear nowhere in pyimzml 1.5.5's parser, so a compressed array
57
+ # is read as ``IMS:1000103 * itemsize`` raw deflate bytes and handed to
58
+ # ``np.frombuffer``, which returns numbers of the correct count.
59
+ _ZLIB_ACCESSION = "MS:1000574"
60
+ _ZLIB_NAME = "zlib compression"
61
+
62
+ # Precision characters whose numpy itemsize is not the itemsize pyimzml seeks
63
+ # with. ``SIZE_DICT['l']`` is 8 on every platform, but ``np.dtype('l').itemsize``
64
+ # is 4 on Windows and 8 on Linux, so pyimzml reads N*8 bytes and decodes 2N
65
+ # int32 values here where it decodes N int64 values there -- one file, two
66
+ # answers. '32-bit integer' ('i') is deliberately NOT in this set: MS:1000519 is
67
+ # spec-legal, pyimzml's own writer emits it for ``intensity_dtype=np.int32``,
68
+ # and it converts correctly today.
69
+ _PLATFORM_DEPENDENT_PRECISIONS = frozenset({"l"})
70
+
71
+
72
+ class _OffsetArrays(NamedTuple):
73
+ """The four per-spectrum offset and length arrays, as int64."""
74
+
75
+ mz_offsets: NDArray[np.int64]
76
+ mz_lengths: NDArray[np.int64]
77
+ int_offsets: NDArray[np.int64]
78
+ int_lengths: NDArray[np.int64]
79
+
80
+
81
+ def _offset_arrays(parser: ImzMLParser) -> _OffsetArrays:
82
+ """Materialise pyimzml's four offset/length lists as int64 arrays.
83
+
84
+ ``np.fromiter(..., count=n)`` rather than ``np.asarray(..., dtype=...)``:
85
+ measured 54.6 ms against 73.5 ms over xenium's 918,855 spectra, and these
86
+ four conversions are most of what validation costs.
87
+
88
+ Args:
89
+ parser: An initialized ImzML parser.
90
+
91
+ Returns:
92
+ The parser's m/z and intensity offsets and lengths.
93
+ """
94
+ n = len(parser.mzOffsets)
95
+ return _OffsetArrays(
96
+ np.fromiter(parser.mzOffsets, dtype=np.int64, count=n),
97
+ np.fromiter(parser.mzLengths, dtype=np.int64, count=n),
98
+ np.fromiter(parser.intensityOffsets, dtype=np.int64, count=n),
99
+ np.fromiter(parser.intensityLengths, dtype=np.int64, count=n),
100
+ )
101
+
102
+
103
+ def _binary_array_specs(
104
+ parser: ImzMLParser,
105
+ ) -> Tuple[Tuple[str, Any, Optional[str]], ...]:
106
+ """Return ``(label, param group id, precision char)`` for both arrays.
107
+
108
+ Args:
109
+ parser: An initialized ImzML parser.
110
+
111
+ Returns:
112
+ One tuple for the m/z array and one for the intensity array.
113
+ """
114
+ return (
115
+ ("m/z", parser.mzGroupId, parser.mzPrecision),
116
+ ("intensity", parser.intGroupId, parser.intensityPrecision),
117
+ )
118
+
119
+
120
+ def _cv_param_int(elem: Any, accession: str) -> Optional[int]:
121
+ """Read one cvParam's value off an element as an int.
122
+
123
+ Args:
124
+ elem: The XML element to search directly beneath.
125
+ accession: The cvParam accession to look for.
126
+
127
+ Returns:
128
+ The integer value, or None if the param is absent or not an integer.
129
+ """
130
+ node = elem.find(f'{_MZML_NS}cvParam[@accession="{accession}"]')
131
+ if node is None:
132
+ return None
133
+ try:
134
+ return int(node.attrib["value"])
135
+ except (KeyError, ValueError):
136
+ return None
137
+
138
+
139
+ def _first_spectrum_array_lengths(
140
+ imzml_path: Path,
141
+ ) -> Dict[Any, Tuple[Optional[int], Optional[int]]]:
142
+ """Read spectrum 0's declared array and encoded lengths, per param group.
143
+
144
+ ``ImzMLParser`` prunes every ``<spectrum>`` out of the tree as it streams
145
+ and keeps only ``IMS:1000102`` and ``IMS:1000103``, so ``IMS:1000104`` --
146
+ the *encoded* byte length, and the only independent witness to the decode
147
+ width -- is gone by the time the parser returns. Re-reading it for the
148
+ first spectrum costs the header plus one spectrum element however large the
149
+ file is.
150
+
151
+ Args:
152
+ imzml_path: Path to the imzML file.
153
+
154
+ Returns:
155
+ A mapping of ``referenceableParamGroupRef`` to
156
+ ``(array_length, encoded_length)``, either of which is None when the
157
+ file does not declare it. Empty if the file declares no spectra.
158
+ """
159
+ # Same document, same stdlib parser pyimzml itself used a moment ago; see
160
+ # the note on the import.
161
+ events = ElementTree.iterparse(str(imzml_path), events=("end",)) # nosec B314
162
+ for _event, elem in events:
163
+ if elem.tag != _MZML_NS + "spectrum":
164
+ continue
165
+ arrays: Dict[Any, Tuple[Optional[int], Optional[int]]] = {}
166
+ for node in elem.findall(
167
+ f"{_MZML_NS}binaryDataArrayList/{_MZML_NS}binaryDataArray"
168
+ ):
169
+ ref_node = node.find(f"{_MZML_NS}referenceableParamGroupRef")
170
+ if ref_node is None:
171
+ continue
172
+ arrays[ref_node.attrib.get("ref")] = (
173
+ _cv_param_int(node, _ARRAY_LENGTH_ACCESSION),
174
+ _cv_param_int(node, _ENCODED_LENGTH_ACCESSION),
175
+ )
176
+ return arrays
177
+ return {}
178
+
41
179
 
42
180
  def _dedupe_sorted(a: NDArray[Any]) -> NDArray[Any]:
43
181
  """Deduplicate an ALREADY-SORTED array, as ``np.unique`` would.
@@ -384,6 +522,11 @@ class ImzMLReader(BaseMSIReader):
384
522
 
385
523
  # Parser initialization flag for lazy loading
386
524
  self._parser_initialized: bool = False
525
+ # And the failure, if it failed. Initialization is expensive -- 63
526
+ # seconds of XML on a 2.1 GB imzML -- and every public entry point
527
+ # calls _ensure_parser_initialized, so a refused file would otherwise
528
+ # be parsed again for each of them before failing the same way.
529
+ self._parser_init_error: Optional[Exception] = None
387
530
 
388
531
  # Cached properties
389
532
  self._common_mass_axis: Optional[NDArray[np.float64]] = None
@@ -396,12 +539,33 @@ class ImzMLReader(BaseMSIReader):
396
539
  self.filepath = data_path
397
540
 
398
541
  def _ensure_parser_initialized(self) -> None:
399
- """Guarantee parser is initialized exactly once."""
400
- if not self._parser_initialized:
401
- if self.filepath is None:
402
- raise ValueError("No file path provided for parser initialization")
542
+ """Guarantee parser is initialized exactly once, success or failure.
543
+
544
+ Raises:
545
+ ValueError: If no file path was given.
546
+ Exception: Whatever the first initialization attempt raised. A file
547
+ that has been refused once is refused from the memo rather than
548
+ parsed again.
549
+ """
550
+ if self._parser_initialized:
551
+ return
552
+ # The stored exception is re-raised as itself rather than rebuilt as
553
+ # ``type(e)(str(e))``, which would lose both the type and the message
554
+ # for any exception whose constructor takes something other than a
555
+ # single string, and would drop the traceback of the attempt that
556
+ # actually failed.
557
+ if self._parser_init_error is not None:
558
+ raise self._parser_init_error
559
+ if self.filepath is None:
560
+ # Not memoized: nothing was parsed, and the caller can still fix it
561
+ # by setting a path.
562
+ raise ValueError("No file path provided for parser initialization")
563
+ try:
403
564
  self._initialize_parser(self.filepath)
404
- self._parser_initialized = True
565
+ except Exception as e:
566
+ self._parser_init_error = e
567
+ raise
568
+ self._parser_initialized = True
405
569
 
406
570
  def _initialize_parser(self, imzml_path: Union[str, Path]) -> None:
407
571
  """Initialize the ImzML parser with the given path.
@@ -446,6 +610,18 @@ class ImzMLReader(BaseMSIReader):
446
610
  # the tree in Python, so there is no C parser state to invalidate;
447
611
  # it is also pyimzml's own default, parses byte-identically, and is
448
612
  # faster here -- 67s vs 118s on a 2.1 GB imzML at the same peak RSS.
613
+ #
614
+ # Two more pyimzml constraints live on this call and neither is
615
+ # visible from it. The parser holds one unsynchronised file handle:
616
+ # get_spectrum_as_string does two seek-then-read pairs on self.m, so
617
+ # sharing a parser across threads returns another pixel's masses at
618
+ # the correct declared length -- 5.4% of reads over 8 threads on
619
+ # real pea, 84 of 108 undetectable. Harmless today because these
620
+ # reads are strictly serial; a trap directly under the obvious
621
+ # parallelisation. And ibd_file= below is not a convenience: without
622
+ # it pyimzml resolves the .ibd itself, by a rule that is worse than
623
+ # this one on case, on directories and on partial downloads.
624
+ # docs/imzml-parser-notes.md has both, with the measurements.
449
625
  self.parser = ImzMLParser(
450
626
  filename=str(imzml_path),
451
627
  parse_lib="ElementTree",
@@ -479,6 +655,310 @@ class ImzMLReader(BaseMSIReader):
479
655
  if self.cache_coordinates:
480
656
  self._cache_all_coordinates()
481
657
 
658
+ try:
659
+ self._validate_parser_state()
660
+ except Exception:
661
+ self.close()
662
+ raise
663
+
664
+ def _validate_parser_state(self) -> None:
665
+ """Check pyimzml's parser state against the .ibd before anything reads.
666
+
667
+ pyimzml seeks and reads unconditionally, and ``np.frombuffer`` objects
668
+ only when the byte count is not a whole multiple of the item size -- so
669
+ an offset pointing past the end of the ``.ibd`` yields an *empty* array
670
+ rather than an error, and the affected pixels leave the store without a
671
+ word. Nothing else in Thyra reads ``mzOffsets``, ``intensityOffsets``,
672
+ ``mzLengths`` or ``intensityLengths``, and nothing else stats the
673
+ ``.ibd``; the only check that exists today is that the file is there.
674
+
675
+ What is refused, in order:
676
+
677
+ 1. ``MS:1000574 zlib compression`` on either binary array. pyimzml
678
+ 1.5.5 has no decompression path at all, so it would hand raw deflate
679
+ bytes to ``np.frombuffer`` and get numbers of the declared length.
680
+ 2. A param group declaring no precision term, more than one, or one
681
+ that disagrees with the precision pyimzml resolved -- pyimzml breaks
682
+ ties by dictionary order rather than by the document. Also
683
+ ``64-bit integer``, whose width is platform-dependent.
684
+ 3. Spectrum 0's ``IMS:1000104`` encoded byte length disagreeing with
685
+ ``IMS:1000103 x itemsize``. This is the only check that catches a
686
+ correct accession carrying a wrong *name*, which makes pyimzml
687
+ decode float64 bytes as float32 at exactly the right length.
688
+ 4. A negative offset or length, or a spectrum whose m/z and intensity
689
+ arrays declare different numbers of values.
690
+ 5. A spectrum whose array ends past the end of the ``.ibd``.
691
+
692
+ Warned about but allowed: non-monotonic offsets, a maximum end byte
693
+ short of the file size, and more than one ``<scanSettings>`` block.
694
+ All three are legal; the last is mishandled downstream (pyimzml
695
+ resolves each scan-settings accession by first match anywhere in the
696
+ list, so a two-block file yields a per-accession chimera), but refusing
697
+ it belongs with the pixel-size unit work rather than here.
698
+
699
+ Limitation: this runs *after* ``ImzMLParser.__fix_offsets``, which
700
+ silently adds 2**32 to every offset from the first positive-to-negative
701
+ transition onward. Seeing the raw offsets would need an override of the
702
+ name-mangled ``_ImzMLParser__fix_offsets``; that buys exactly one case
703
+ these checks miss -- a spurious negative on the very last spectrum,
704
+ where the repair leaves no later read to push past the end of the file
705
+ -- and roughly doubles what validation costs. The repair is dormant on
706
+ every file measured (0 negatives and 0 inversions in 1,896,000 offsets
707
+ across bellini, pea and xenium), so it is left alone.
708
+
709
+ Raises:
710
+ ValueError: If the parser's state cannot produce correct reads.
711
+ """
712
+ parser = cast(ImzMLParser, self.parser)
713
+
714
+ for label, group_id, precision in _binary_array_specs(parser):
715
+ self._validate_binary_group(parser, label, group_id, precision)
716
+
717
+ self._validate_encoded_lengths(parser)
718
+
719
+ arrays = _offset_arrays(parser)
720
+ self._validate_offset_arrays(arrays)
721
+ self._validate_ibd_extent(parser, arrays)
722
+
723
+ n_scan_settings = len(parser.metadata.scan_settings)
724
+ if n_scan_settings != 1:
725
+ logger.warning(
726
+ f"imzML declares {n_scan_settings} <scanSettings> blocks. "
727
+ "pyimzml resolves each scan-settings accession by first match "
728
+ "anywhere in the list, so the pixel size and pixel counts "
729
+ "Thyra reads may come from different blocks and describe no "
730
+ "single region."
731
+ )
732
+
733
+ def _validate_binary_group(
734
+ self,
735
+ parser: ImzMLParser,
736
+ label: str,
737
+ group_id: Any,
738
+ precision: Optional[str],
739
+ ) -> None:
740
+ """Check the referenceable param group behind one binary array.
741
+
742
+ Args:
743
+ parser: An initialized ImzML parser.
744
+ label: ``"m/z"`` or ``"intensity"``, for the messages.
745
+ group_id: The param group id pyimzml resolved for this array.
746
+ precision: The precision character pyimzml resolved for it.
747
+
748
+ Raises:
749
+ ValueError: If the group is missing, declares zlib compression,
750
+ declares no precision term or more than one, disagrees with the
751
+ precision pyimzml resolved, or names a type whose width is
752
+ platform-dependent.
753
+ """
754
+ groups = parser.metadata.referenceable_param_groups
755
+ group = groups.get(group_id)
756
+ if group is None or precision is None:
757
+ raise ValueError(
758
+ f"imzML declares no usable referenceable param group for the "
759
+ f"{label} array, so pyimzml cannot know how to decode it "
760
+ f"(looked for {group_id!r} among {sorted(map(str, groups))})."
761
+ )
762
+
763
+ if _ZLIB_ACCESSION in group or _ZLIB_NAME in group:
764
+ raise ValueError(
765
+ f"imzML declares zlib compression ({_ZLIB_ACCESSION}) on its "
766
+ f"{label} array. pyimzml 1.5.5 has no decompression path: it "
767
+ f"reads IMS:1000103 x itemsize raw deflate bytes and decodes "
768
+ f"them as numbers, which succeeds silently at the declared "
769
+ f"length. Re-export with MS:1000576 no compression."
770
+ )
771
+
772
+ # param_by_name, not the group's own cv_params: declaring the precision
773
+ # in a param group this one *references* is legal, and reading
774
+ # cv_params would refuse that file.
775
+ declared = [
776
+ name for name in parser.precisionDict if name in group.param_by_name
777
+ ]
778
+ if len(declared) != 1:
779
+ raise ValueError(
780
+ f"imzML param group {group_id!r} declares {len(declared)} "
781
+ f"precision terms for the {label} array "
782
+ f"({', '.join(declared) if declared else 'none'}); exactly one "
783
+ f"is required. pyimzml breaks ties by dictionary order rather "
784
+ f"than by the document, so the decode width would be arbitrary."
785
+ )
786
+
787
+ resolved = parser.precisionDict[declared[0]]
788
+ if resolved != precision:
789
+ raise ValueError(
790
+ f"imzML param group {group_id!r} declares {declared[0]!r} for "
791
+ f"the {label} array, which is {resolved!r}, but pyimzml "
792
+ f"resolved {precision!r} and will decode at that width."
793
+ )
794
+
795
+ if precision in _PLATFORM_DEPENDENT_PRECISIONS:
796
+ raise ValueError(
797
+ f"imzML declares {declared[0]!r} for its {label} array. "
798
+ f"pyimzml reads {parser.sizeDict[precision]} bytes per value "
799
+ f"but decodes them as numpy's 'l', whose itemsize is "
800
+ f"{np.dtype(precision).itemsize} on this platform, so the same "
801
+ f"file reads differently on Windows and on Linux. Re-export "
802
+ f"the array as 32-bit or 64-bit float."
803
+ )
804
+
805
+ def _validate_encoded_lengths(self, parser: ImzMLParser) -> None:
806
+ """Cross-check spectrum 0's IMS:1000104 against IMS:1000103 x itemsize.
807
+
808
+ pyimzml's ``ACCESSION_FIX_MAPPING`` rewrites the *accession* of a
809
+ mis-declared 32/64-bit float term and keeps the raw *name*, and the
810
+ precision is then derived from the name -- so a correct ``MS:1000523``
811
+ carrying the name ``64-bit float`` makes every m/z value in the file
812
+ decode at the wrong width, at exactly the declared length, announced
813
+ only by a ``UserWarning`` worded as a successful repair. ``IMS:1000104``
814
+ is the independent witness: it equalled ``IMS:1000103 x itemsize`` for
815
+ all ~1.9M arrays across bellini, pea and xenium, 0 violations.
816
+
817
+ Checked on spectrum 0 only. Every spectrum would mean re-reading the
818
+ whole document.
819
+
820
+ Args:
821
+ parser: An initialized ImzML parser.
822
+
823
+ Raises:
824
+ ValueError: If a declared encoded length contradicts the precision
825
+ pyimzml resolved.
826
+ """
827
+ if self.imzml_path is None:
828
+ return
829
+ try:
830
+ arrays = _first_spectrum_array_lengths(self.imzml_path)
831
+ except ElementTree.ParseError as e:
832
+ # pyimzml has already parsed this document successfully, so a
833
+ # failure here is this function's problem and must not condemn the
834
+ # file.
835
+ logger.debug(
836
+ f"Could not re-read spectrum 0 for the "
837
+ f"{_ENCODED_LENGTH_ACCESSION} cross-check: {e}"
838
+ )
839
+ return
840
+
841
+ for label, group_id, precision in _binary_array_specs(parser):
842
+ declared = arrays.get(group_id)
843
+ if declared is None:
844
+ continue
845
+ array_length, encoded_length = declared
846
+ if array_length is None or encoded_length is None:
847
+ continue
848
+ itemsize = parser.sizeDict[precision]
849
+ expected = array_length * itemsize
850
+ if encoded_length != expected:
851
+ raise ValueError(
852
+ f"imzML spectrum 0 declares {encoded_length:,} encoded "
853
+ f"bytes ({_ENCODED_LENGTH_ACCESSION}) for its {label} "
854
+ f"array, but its {array_length:,} values at the resolved "
855
+ f"precision {precision!r} occupy {expected:,} bytes "
856
+ f"({_ARRAY_LENGTH_ACCESSION} x {itemsize}). pyimzml reads "
857
+ f"{expected:,} bytes, so the array decodes at the wrong "
858
+ f"width or the wrong length."
859
+ )
860
+
861
+ def _validate_offset_arrays(self, arrays: _OffsetArrays) -> None:
862
+ """Check the offset and length arrays for internally impossible values.
863
+
864
+ Note that ``mz_len == int_len`` is not a corruption guard: the lengths
865
+ come from ``IMS:1000103`` while pyimzml's offset repair touches only
866
+ ``IMS:1000102``, so a re-pointed read agrees with itself. It is here
867
+ because it is nearly free and it does catch truncation faults.
868
+
869
+ Args:
870
+ arrays: The parser's offsets and lengths, as int64.
871
+
872
+ Raises:
873
+ ValueError: If any value is negative, or if a spectrum's m/z and
874
+ intensity arrays declare different numbers of values.
875
+ """
876
+ for label, values in (
877
+ ("m/z offset", arrays.mz_offsets),
878
+ ("m/z length", arrays.mz_lengths),
879
+ ("intensity offset", arrays.int_offsets),
880
+ ("intensity length", arrays.int_lengths),
881
+ ):
882
+ negative = np.flatnonzero(values < 0)
883
+ if negative.size:
884
+ idx = int(negative[0])
885
+ raise ValueError(
886
+ f"imzML spectrum {idx} declares a negative {label} of "
887
+ f"{int(values[idx]):,} ({negative.size:,} spectra "
888
+ f"affected). Offsets and lengths are byte and element "
889
+ f"counts and cannot be negative; pyimzml's signed-32-bit "
890
+ f"offset repair did not remove this one."
891
+ )
892
+
893
+ mismatch = np.flatnonzero(arrays.mz_lengths != arrays.int_lengths)
894
+ if mismatch.size:
895
+ idx = int(mismatch[0])
896
+ raise ValueError(
897
+ f"imzML spectrum {idx} declares {int(arrays.mz_lengths[idx]):,} "
898
+ f"m/z values but {int(arrays.int_lengths[idx]):,} intensity "
899
+ f"values ({mismatch.size:,} spectra disagree). A spectrum's two "
900
+ f"arrays describe the same peaks and must be the same length."
901
+ )
902
+
903
+ def _validate_ibd_extent(self, parser: ImzMLParser, arrays: _OffsetArrays) -> None:
904
+ """Check that every declared array lies inside the .ibd.
905
+
906
+ Args:
907
+ parser: An initialized ImzML parser.
908
+ arrays: The parser's offsets and lengths, as int64.
909
+
910
+ Raises:
911
+ ValueError: If any spectrum's array ends past the end of the
912
+ ``.ibd``.
913
+ """
914
+ if self.ibd_path is None or arrays.mz_offsets.size == 0:
915
+ return
916
+ ibd_size = self.ibd_path.stat().st_size
917
+
918
+ max_end = 0
919
+ for label, offsets, lengths, precision in (
920
+ ("m/z", arrays.mz_offsets, arrays.mz_lengths, parser.mzPrecision),
921
+ (
922
+ "intensity",
923
+ arrays.int_offsets,
924
+ arrays.int_lengths,
925
+ parser.intensityPrecision,
926
+ ),
927
+ ):
928
+ end = offsets + lengths * parser.sizeDict[precision]
929
+ max_end = max(max_end, int(end.max()))
930
+ past = np.flatnonzero(end > ibd_size)
931
+ if past.size:
932
+ idx = int(past[0])
933
+ raise ValueError(
934
+ f"imzML spectrum {idx} declares a {label} array ending at "
935
+ f"byte {int(end[idx]):,}, but {self.ibd_path.name} is "
936
+ f"{ibd_size:,} bytes ({past.size:,} spectra are affected; "
937
+ f"the furthest ends at {int(end.max()):,}). The .ibd is "
938
+ f"truncated or its offsets are wrong -- pyimzml would "
939
+ f"return empty arrays for these spectra without raising, "
940
+ f"and they would simply be missing from the output."
941
+ )
942
+
943
+ if max_end != ibd_size:
944
+ logger.warning(
945
+ f"{self.ibd_path.name} is {ibd_size:,} bytes but the last byte "
946
+ f"any spectrum declares is {max_end:,}, leaving "
947
+ f"{ibd_size - max_end:,} unaccounted for. Trailing bytes are "
948
+ f"legal, but this is also what a partially-copied .ibd looks "
949
+ f"like."
950
+ )
951
+
952
+ if np.any(np.diff(arrays.mz_offsets) < 0) or np.any(
953
+ np.diff(arrays.int_offsets) < 0
954
+ ):
955
+ logger.warning(
956
+ "imzML offsets are not monotonically non-decreasing. This is "
957
+ "legal, but pyimzml's signed-32-bit offset repair assumes "
958
+ "document order matches byte order and silently rewrites every "
959
+ "offset after a sign flip when it does not."
960
+ )
961
+
482
962
  def _cache_all_coordinates(self) -> None:
483
963
  """Cache all coordinates for faster access.
484
964
 
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes