tiny-metaio 0.3.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/PKG-INFO +25 -4
  2. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/README.md +24 -3
  3. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/pyproject.toml +1 -1
  4. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/metaio/__main__.py +6 -4
  5. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/metaio/image.py +22 -13
  6. tiny_metaio-0.4.0/src/metaio/pack.py +291 -0
  7. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/metaio/quant.py +13 -5
  8. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/tiny_metaio.egg-info/PKG-INFO +25 -4
  9. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/tiny_metaio.egg-info/SOURCES.txt +3 -0
  10. tiny_metaio-0.4.0/src/tiny_metaio.egg-info/scm_file_list.json +24 -0
  11. tiny_metaio-0.4.0/src/tiny_metaio.egg-info/scm_version.json +8 -0
  12. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/tests/test_quantization.py +61 -5
  13. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/.gitignore +0 -0
  14. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/.gitlab-ci.yml +0 -0
  15. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/.pre-commit-config.yaml +0 -0
  16. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/Containerfile +0 -0
  17. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/setup.cfg +0 -0
  18. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/metaio/__init__.py +0 -0
  19. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/metaio/mha.py +0 -0
  20. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/metaio/nifti.py +0 -0
  21. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/metaio/py.typed +0 -0
  22. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/metaio/write.py +0 -0
  23. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/tiny_metaio.egg-info/dependency_links.txt +0 -0
  24. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/tiny_metaio.egg-info/entry_points.txt +0 -0
  25. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/tiny_metaio.egg-info/requires.txt +0 -0
  26. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/src/tiny_metaio.egg-info/top_level.txt +0 -0
  27. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/tests/conftest.py +0 -0
  28. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/tests/test_symmetry.py +0 -0
  29. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/tests/test_vs_simpleitk.py +0 -0
  30. {tiny_metaio-0.3.0 → tiny_metaio-0.4.0}/uv.lock +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tiny-metaio
3
- Version: 0.3.0
3
+ Version: 0.4.0
4
4
  Summary: Read and write MetaImages with minimal dependencies
5
5
  Author-email: Nicolas Cedilnik <nicolas.cedilnik@inria.fr>
6
6
  Project-URL: Homepage, https://gitlab.inria.fr/ncedilni/metaio
@@ -26,7 +26,7 @@ images (`.nii` or `.nii.gz`)
26
26
  and write `.mha` images
27
27
  in python with minimal dependencies.
28
28
 
29
- Bonus: provides a custom 8-bit MHA-like format when size matters: `.metaq`.
29
+ Bonus: provides a custom MHA-like format when size matters: `.metaq`.
30
30
 
31
31
  ## Installation
32
32
 
@@ -61,6 +61,27 @@ array([0., 0.])
61
61
 
62
62
  ### Custom format
63
63
 
64
+ #### Integers
65
+
66
+ Integers are stored without loss in a very compact way.
67
+
68
+ ```python
69
+ >>> from pathlib import Path
70
+ >>> import numpy as np
71
+ >>> img = metaio.MetaImage([[-5] * 1000, [2] * 1000])
72
+ >>> img.save("some-name.mha")
73
+ >>> Path("some-name.mha").stat().st_size
74
+ 16246
75
+ >>> img.save_compact("some-name.metaq")
76
+ >>> round(Path("some-name.metaq").stat().st_size / Path("some-name.mha").stat().st_size, 2)
77
+ 0.09
78
+ >>> np.all(metaio.read("some-name.metaq").data == img.data)
79
+ np.True_
80
+
81
+ ```
82
+
83
+ #### Floats
84
+
64
85
  You need to specify a range of values covered by the quantization.
65
86
  Values outside this ranges will be clipped.
66
87
  A loss of precision is expected.
@@ -68,7 +89,7 @@ NaNs will be implicitly converted to zero.
68
89
 
69
90
  ```python
70
91
  >>> img = metaio.MetaImage([[-0.5, 1], [2.5, 5]])
71
- >>> img.save_quantized("some-name.metaq", min_=0, max_=2.5)
92
+ >>> img.save_compact("some-name.metaq", min_=0, max_=2.5)
72
93
  >>> metaio.read("some-name.metaq").data
73
94
  array([[0. , 1. ],
74
95
  [2.5, 2.5]], dtype=float32)
@@ -81,7 +102,7 @@ Values where the mask is `False` will not be stored at all and restored as NaN o
81
102
 
82
103
  ```python
83
104
  >>> img = metaio.MetaImage([[-0.5, 1], [2.5, 5]])
84
- >>> img.save_quantized("some-name.metaq", min_=0, max_=2.5, mask=img.data <= 2.5)
105
+ >>> img.save_compact("some-name.metaq", min_=0, max_=2.5, mask=img.data <= 2.5)
85
106
  >>> metaio.read("some-name.metaq").data
86
107
  array([[0. , 1. ],
87
108
  [2.5, nan]], dtype=float32)
@@ -9,7 +9,7 @@ images (`.nii` or `.nii.gz`)
9
9
  and write `.mha` images
10
10
  in python with minimal dependencies.
11
11
 
12
- Bonus: provides a custom 8-bit MHA-like format when size matters: `.metaq`.
12
+ Bonus: provides a custom MHA-like format when size matters: `.metaq`.
13
13
 
14
14
  ## Installation
15
15
 
@@ -44,6 +44,27 @@ array([0., 0.])
44
44
 
45
45
  ### Custom format
46
46
 
47
+ #### Integers
48
+
49
+ Integers are stored without loss in a very compact way.
50
+
51
+ ```python
52
+ >>> from pathlib import Path
53
+ >>> import numpy as np
54
+ >>> img = metaio.MetaImage([[-5] * 1000, [2] * 1000])
55
+ >>> img.save("some-name.mha")
56
+ >>> Path("some-name.mha").stat().st_size
57
+ 16246
58
+ >>> img.save_compact("some-name.metaq")
59
+ >>> round(Path("some-name.metaq").stat().st_size / Path("some-name.mha").stat().st_size, 2)
60
+ 0.09
61
+ >>> np.all(metaio.read("some-name.metaq").data == img.data)
62
+ np.True_
63
+
64
+ ```
65
+
66
+ #### Floats
67
+
47
68
  You need to specify a range of values covered by the quantization.
48
69
  Values outside this ranges will be clipped.
49
70
  A loss of precision is expected.
@@ -51,7 +72,7 @@ NaNs will be implicitly converted to zero.
51
72
 
52
73
  ```python
53
74
  >>> img = metaio.MetaImage([[-0.5, 1], [2.5, 5]])
54
- >>> img.save_quantized("some-name.metaq", min_=0, max_=2.5)
75
+ >>> img.save_compact("some-name.metaq", min_=0, max_=2.5)
55
76
  >>> metaio.read("some-name.metaq").data
56
77
  array([[0. , 1. ],
57
78
  [2.5, 2.5]], dtype=float32)
@@ -64,7 +85,7 @@ Values where the mask is `False` will not be stored at all and restored as NaN o
64
85
 
65
86
  ```python
66
87
  >>> img = metaio.MetaImage([[-0.5, 1], [2.5, 5]])
67
- >>> img.save_quantized("some-name.metaq", min_=0, max_=2.5, mask=img.data <= 2.5)
88
+ >>> img.save_compact("some-name.metaq", min_=0, max_=2.5, mask=img.data <= 2.5)
68
89
  >>> metaio.read("some-name.metaq").data
69
90
  array([[0. , 1. ],
70
91
  [2.5, nan]], dtype=float32)
@@ -58,7 +58,7 @@ ruff = "ruff check"
58
58
  typos = "typos"
59
59
  lint = ["ty", "ruff", "typos"]
60
60
  autofix = "ruff check --fix"
61
- test = "coverage run -m pytest --doctest-glob='*.md' --junitxml=report.xml"
61
+ test = "coverage run -m pytest --doctest-modules --doctest-glob='*.md' --junitxml=report.xml"
62
62
  coverage-report = "coverage report"
63
63
  coverage-xml = "coverage xml"
64
64
  coverage-html = "coverage html"
@@ -14,25 +14,27 @@ def main() -> int:
14
14
  )
15
15
  parser.add_argument(
16
16
  "--no-compression",
17
- help="Disable compression of the output. Only effective for '.mha'.",
17
+ help="Disable compression of the output. Only used for '.mha'.",
18
18
  dest="compression",
19
19
  action="store_false",
20
20
  )
21
21
  parser.add_argument(
22
22
  "--quantization-range",
23
- help="Quantization range. Only effective for '.metaq'.",
23
+ help="Quantization range. Only used for '.metaq', and only for floating point data types.",
24
24
  type=float,
25
25
  nargs=2,
26
26
  default=(None, None),
27
27
  )
28
28
  parser.add_argument(
29
- "--mask", help="Boolean mask to use. Only effective for '.metaq'.", type=Path
29
+ "--mask",
30
+ help="Boolean mask to use. Only used for '.metaq' and floating point data types.",
31
+ type=Path,
30
32
  )
31
33
  args = parser.parse_args()
32
34
  img = read(args.INPUT)
33
35
  if args.OUTPUT.suffix == ".metaq":
34
36
  mask = None if args.mask is None else read(args.mask).data.astype(bool)
35
- img.save_quantized(args.OUTPUT, *args.quantization_range, mask=mask)
37
+ img.save_compact(args.OUTPUT, *args.quantization_range, mask=mask)
36
38
  else:
37
39
  img.save(args.OUTPUT, compress=args.compression)
38
40
  return 0
@@ -2,12 +2,14 @@ from __future__ import annotations
2
2
 
3
3
  import shutil
4
4
  import zlib
5
+ from collections.abc import Mapping
5
6
  from pathlib import Path
7
+ from typing import Any
6
8
 
7
9
  import numpy as np
8
10
  import numpy.typing as npt
9
11
 
10
- from . import write
12
+ from . import pack, write
11
13
 
12
14
 
13
15
  class MetaImage:
@@ -135,7 +137,7 @@ class MetaImage:
135
137
 
136
138
  hfh.write(pixel_bytes)
137
139
 
138
- def save_quantized(
140
+ def save_compact(
139
141
  self,
140
142
  path: Path | str,
141
143
  min_: float | None = None,
@@ -149,26 +151,33 @@ class MetaImage:
149
151
  "For quantized images, the file extension must be '.metaq'", path.name
150
152
  )
151
153
  if mask is None:
152
- kwargs = {}
154
+ kwargs: Mapping[str, npt.NDArray[Any] | float | int] = {}
153
155
  data = self.data.astype(np.float32)
154
156
  else:
155
157
  kwargs = {"mask": np.packbits(mask.astype(bool)), "mask_shape": mask.shape}
156
158
  data = self.data[mask].astype(np.float32)
157
- if min_ is None:
158
- min_ = np.nanmin(self.data)
159
- if max_ is None:
160
- max_ = np.nanmax(self.data)
161
- quantized = _quantize(data, min_, max_)
162
- meta = _dict_to_structured_array(self.metadata)
159
+ if np.issubdtype(self.data.dtype, np.floating):
160
+ if min_ is None:
161
+ min_ = np.nanmin(self.data)
162
+ if max_ is None:
163
+ max_ = np.nanmax(self.data)
164
+ kwargs["data"] = _quantize(data, min_, max_)
165
+ kwargs["min"] = min_ # ty:ignore
166
+ kwargs["max"] = max_ # ty:ignore
167
+ elif np.issubdtype(self.data.dtype, np.integer):
168
+ if min_ is not None or max_ is not None:
169
+ raise TypeError("min and max should only be used for float datatypes")
170
+ kwargs["packed"] = pack.pack_compact(self.data)
171
+ kwargs["packed_shape"] = self.data.shape
172
+ kwargs["packed_dtype"] = self.data.dtype.str
173
+ else:
174
+ raise ValueError
163
175
  np.savez_compressed(
164
176
  path,
165
- data=quantized,
166
177
  spacing=self.spacing,
167
178
  direction=self.direction,
168
179
  origin=self.origin,
169
- min=min_,
170
- max=max_,
171
- meta=meta,
180
+ meta=_dict_to_structured_array(self.metadata),
172
181
  **kwargs,
173
182
  allow_pickle=False,
174
183
  )
@@ -0,0 +1,291 @@
1
+ """
2
+ Pack/unpack arrays of unsigned integers into a dense x-bit stream,
3
+ stored in a uint8 byte array -- like numpy.packbits, but for any width.
4
+ """
5
+
6
+ import numpy as np
7
+ import numpy.typing as npt
8
+
9
+ _IS_LITTLE_ENDIAN = np.array([1], dtype=np.uint16).view(np.uint8)[0] == 1
10
+
11
+
12
+ def packbits_x(a: npt.ArrayLike, width: int) -> npt.NDArray[np.uint8]:
13
+ """
14
+ Pack an array of unsigned integers into a dense `width`-bit stream.
15
+
16
+ Parameters
17
+ ----------
18
+ a : array_like of non-negative integers, each value < 2**width
19
+ width : int, bits per element (1-32)
20
+
21
+ Returns
22
+ -------
23
+ packed : npt.NDArray[np.uint8]
24
+ Bit-packed bytes; length = ceil(len(a) * width / 8).
25
+ Bit order is big-endian (MSB first within each element).
26
+
27
+ Examples
28
+ --------
29
+ >>> packbits_x([0b000000, 0b111111, 0b000001, 0b111110], width=6)
30
+ array([ 3, 240, 126], dtype=uint8)
31
+
32
+ # 000000 111111 000001 111110 -> 00000011 11110000 01111110
33
+ """
34
+ # uint64 is required: the left-shift may exceed 32 bits, and .view(uint8).reshape(n,8)
35
+ # assumes exactly 8 bytes per element.
36
+ a = np.asarray(a, dtype=np.uint64).ravel()
37
+ if not (1 <= width <= 32):
38
+ raise ValueError(f"width must be 1-32, got {width}")
39
+ if np.any(a >= (np.uint64(1) << np.uint64(width))):
40
+ raise ValueError(
41
+ f"all values must be < {1 << width} (i.e. fit in {width} bits)"
42
+ )
43
+
44
+ n = len(a)
45
+ padded_width = int(np.ceil(width / 8)) * 8 # round width up to whole bytes
46
+ nbytes_per = padded_width // 8
47
+
48
+ # Left-align each value into the top `width` bits of a padded_width-wide field,
49
+ # then view the backing memory as individual bytes.
50
+ # On little-endian machines the bytes are reversed; flip to get big-endian order
51
+ # so that np.unpackbits yields MSB-first bits.
52
+ shifted = (a << np.uint64(padded_width - width)).view(np.uint8).reshape(n, 8)
53
+ be_bytes = (
54
+ shifted[:, :nbytes_per][:, ::-1]
55
+ if _IS_LITTLE_ENDIAN
56
+ else shifted[:, :nbytes_per]
57
+ )
58
+
59
+ # Unpack all bits at once -> (n, padded_width), trim to (n, width), stream & repack
60
+ bits = np.unpackbits(be_bytes.reshape(-1)).reshape(n, padded_width)[:, :width]
61
+ flat = bits.ravel()
62
+ pad = (-len(flat)) % 8
63
+ if pad:
64
+ flat = np.concatenate([flat, np.zeros(pad, np.uint8)])
65
+ return np.packbits(flat)
66
+
67
+
68
+ def unpackbits_x(
69
+ packed: npt.NDArray[np.uint8],
70
+ width: int,
71
+ count: int | None = None,
72
+ ) -> npt.NDArray[np.unsignedinteger]:
73
+ """
74
+ Unpack a uint8 byte array produced by :func:`packbits_x` back into integers.
75
+
76
+ Parameters
77
+ ----------
78
+ packed : npt.NDArray[np.uint8], 1-D
79
+ width : int, bits per element (must match the width used to pack)
80
+ count : int | None, optional -- number of elements; defaults to all complete elements
81
+
82
+ Returns
83
+ -------
84
+ out : npt.NDArray[np.unsignedinteger]
85
+ dtype is uint8 / uint16 / uint32 / uint64, whichever is smallest for ``width``.
86
+
87
+ Examples
88
+ --------
89
+ >>> unpackbits_x(packbits_x([0, 63, 1, 62], 6), 6, count=4)
90
+ array([ 0, 63, 1, 62], dtype=uint8)
91
+ """
92
+ if not isinstance(packed, np.ndarray):
93
+ raise TypeError(
94
+ f"packed must be a 1-D ndarray of uint8, got {type(packed).__name__}"
95
+ )
96
+ if packed.dtype != np.uint8:
97
+ raise TypeError(f"packed must have dtype uint8, got {packed.dtype}")
98
+ if packed.ndim != 1:
99
+ raise TypeError(f"packed must be 1-D, got shape {packed.shape}")
100
+ if not (1 <= width <= 32):
101
+ raise ValueError(f"width must be 1-32, got {width}")
102
+
103
+ max_count = (len(packed) * 8) // width
104
+ if count is None:
105
+ count = max_count
106
+ elif count > max_count:
107
+ raise ValueError(f"count={count} exceeds available elements ({max_count})")
108
+
109
+ dtype = next(
110
+ dt
111
+ for dt in (np.uint8, np.uint16, np.uint32, np.uint64)
112
+ if np.iinfo(dt).bits >= width
113
+ )
114
+ bits = np.unpackbits(packed)[: count * width].reshape(count, width)
115
+ # Place-value vector [2^(width-1), ..., 2^1, 2^0] matching the MSB-first bit columns.
116
+ # dtype(1) seeds the shift in the output type to avoid numpy upcasting to int64.
117
+ # Row-wise dot product with bits gives the integer value: [1,0,1,1] . [8,4,2,1] = 11.
118
+ powers = dtype(1) << np.arange(width - 1, -1, -1, dtype=dtype)
119
+ return (bits.astype(dtype) * powers).sum(axis=1).astype(dtype)
120
+
121
+
122
+ # Header layout (all big-endian):
123
+ # [0] : magic = 0xB1 (1 byte)
124
+ # [1] : index_bits = bits per index into the unique-value table (1 byte)
125
+ # [2:6] : count = number of elements (uint32, 4 bytes)
126
+ # [6:8] : n_unique = number of unique values (uint16, 2 bytes)
127
+ # [8] : signed = 1 if table values are int64, 0 if uint64 (1 byte)
128
+ # [9:16] : reserved = 7 zero bytes (alignment padding)
129
+ # [16:] : table = sorted unique values, each stored as int64 or uint64 (8 bytes each)
130
+ # Payload (after header): bit-packed indices, index_bits bits each.
131
+ #
132
+ # Strategy: encode each element as its rank in the sorted unique-value table,
133
+ # then pack those indices. [0, 5, 15799] has 3 unique values -> indices need
134
+ # only 2 bits each, regardless of how large the actual values are.
135
+ # Signed arrays are handled identically -- the table just uses int64 instead.
136
+ _MAGIC = np.uint8(0xB1)
137
+ _HEADER_BASE_DTYPE = np.dtype(
138
+ [
139
+ ("magic", ">u1"), # 1 byte
140
+ ("index_bits", ">u1"), # 1 byte
141
+ ("count", ">u4"), # 4 bytes
142
+ ("n_unique", ">u2"), # 2 bytes
143
+ ("signed", ">u1"), # 1 byte -- 1 = int64 table, 0 = uint64 table
144
+ ("reserved", ">u1", (7,)), # 7 bytes padding
145
+ ]
146
+ ) # 16 bytes before the table
147
+
148
+
149
+ def pack_compact(a: npt.ArrayLike) -> npt.NDArray[np.uint8]:
150
+ """
151
+ Pack an integer array into the most compact representation possible.
152
+
153
+ Each element is stored as its index into a sorted table of unique values.
154
+ The number of index bits is `ceil(log2(n_unique))`, so an array whose
155
+ values come from a small alphabet is compressed far more than raw
156
+ bit-packing would allow. Example: `[0, 5, 15799`` repeated 1000 times
157
+ needs only 2 bits per element instead of 14. Signed integers are fully
158
+ supported: the table is stored as int64 when the input contains negative
159
+ values.
160
+
161
+ The output is fully self-describing: the unique-value table and all
162
+ metadata are stored in the header, so `unpack_compact` needs no
163
+ extra arguments.
164
+
165
+ Header layout (big-endian):
166
+ byte 0 : magic 0xB1
167
+ byte 1 : index_bits (bits per index, 1-32)
168
+ bytes 2-5 : count (uint32, number of elements)
169
+ bytes 6-7 : n_unique (uint16, number of unique values)
170
+ byte 8 : signed flag (1 = int64 table, 0 = uint64 table)
171
+ bytes 9-15 : reserved (zero)
172
+ bytes 16+ : table of n_unique int64/uint64 values (8 bytes each, sorted)
173
+ Payload (after header): bit-packed indices, index_bits bits per element.
174
+
175
+ Examples
176
+ --------
177
+ >>> unpack_compact(pack_compact([0, 5, 15799, 5, 0]))
178
+ array([ 0, 5, 15799, 5, 0], dtype=uint16)
179
+ >>> unpack_compact(pack_compact([-3, 0, 1, -3, 1]))
180
+ array([-3, 0, 1, -3, 1], dtype=int8)
181
+ """
182
+ a = np.asarray(a).ravel()
183
+ # Decide signedness from dtype or presence of negative values
184
+
185
+ if len(a) == 0:
186
+ raise ValueError("input array must not be empty")
187
+ if len(a) > np.iinfo(np.uint32).max:
188
+ raise ValueError(f"array too long for header uint32 count: {len(a)}")
189
+
190
+ # Build sorted unique-value table and replace each element with its index
191
+ unique_vals, indices = np.unique(a, return_inverse=True)
192
+ n_unique = len(unique_vals)
193
+ if n_unique > np.iinfo(np.uint16).max:
194
+ raise ValueError(f"too many unique values for header uint16: {n_unique}")
195
+ is_signed = bool(np.any(unique_vals < 0))
196
+
197
+ # Bits needed to represent indices 0..n_unique-1
198
+ index_bits = max(1, (n_unique - 1).bit_length())
199
+
200
+ header = np.zeros(1, dtype=_HEADER_BASE_DTYPE)
201
+ header["magic"] = _MAGIC # ty:ignore
202
+ header["index_bits"] = index_bits # ty:ignore
203
+ header["count"] = len(a) # ty:ignore
204
+ header["n_unique"] = n_unique # ty:ignore
205
+ header["signed"] = np.uint8(is_signed) # ty:ignore
206
+
207
+ # Store table as int64 or uint64; view as uint8 for concatenation
208
+ table = unique_vals.astype(np.int64 if is_signed else np.uint64)
209
+
210
+ return np.concatenate(
211
+ [
212
+ header.view(np.uint8),
213
+ table.view(np.uint8),
214
+ packbits_x(indices.astype(np.uint64), index_bits),
215
+ ]
216
+ )
217
+
218
+
219
+ def unpack_compact(blob: npt.NDArray[np.uint8]) -> npt.NDArray[np.integer]:
220
+ """
221
+ Unpack a uint8 array produced by :func:`pack_compact`.
222
+
223
+ Parameters
224
+ ----------
225
+ blob : npt.NDArray[np.uint8], 1-D
226
+
227
+ Returns
228
+ -------
229
+ out : npt.NDArray[np.integer]
230
+ Values restored from the unique-value table; dtype is the smallest
231
+ signed or unsigned type that fits the range of values in the table.
232
+
233
+ Examples
234
+ --------
235
+ >>> unpack_compact(pack_compact([0, 5, 15799, 5, 0]))
236
+ array([ 0, 5, 15799, 5, 0], dtype=uint16)
237
+ >>> unpack_compact(pack_compact([-3, 0, 1, -3, 1]))
238
+ array([-3, 0, 1, -3, 1], dtype=int8)
239
+ """
240
+ if not isinstance(blob, np.ndarray):
241
+ raise TypeError(
242
+ f"blob must be a 1-D ndarray of uint8, got {type(blob).__name__}"
243
+ )
244
+ if blob.dtype != np.uint8:
245
+ raise TypeError(f"blob must have dtype uint8, got {blob.dtype}")
246
+ if blob.ndim != 1:
247
+ raise TypeError(f"blob must be 1-D, got shape {blob.shape}")
248
+
249
+ base_size = _HEADER_BASE_DTYPE.itemsize # 16 bytes
250
+ if len(blob) < base_size:
251
+ raise ValueError(
252
+ f"blob too short to contain header ({len(blob)} < {base_size} bytes)"
253
+ )
254
+
255
+ header = blob[:base_size].view(_HEADER_BASE_DTYPE)[0]
256
+ if header["magic"] != _MAGIC:
257
+ raise ValueError(
258
+ f"invalid magic byte: expected 0x{_MAGIC:02X}, got 0x{int(header['magic']):02X}"
259
+ )
260
+
261
+ index_bits = int(header["index_bits"])
262
+ count = int(header["count"])
263
+ n_unique = int(header["n_unique"])
264
+ is_signed = bool(header["signed"])
265
+ table_bytes = n_unique * 8 # each value is 8 bytes (int64 or uint64)
266
+
267
+ header_size = base_size + table_bytes
268
+ if len(blob) < header_size:
269
+ raise ValueError(
270
+ f"blob too short to contain table ({len(blob)} < {header_size} bytes)"
271
+ )
272
+
273
+ table_dtype = np.int64 if is_signed else np.uint64
274
+ table = blob[base_size:header_size].view(table_dtype)
275
+ indices = unpackbits_x(blob[header_size:], index_bits, count=count)
276
+
277
+ # choose the smallest dtype that fits the full range of the table
278
+ lo, hi = int(table.min()), int(table.max())
279
+ if is_signed:
280
+ dtype = next(
281
+ dt
282
+ for dt in (np.int8, np.int16, np.int32, np.int64)
283
+ if np.iinfo(dt).min <= lo and np.iinfo(dt).max >= hi
284
+ )
285
+ else:
286
+ dtype = next(
287
+ dt
288
+ for dt in (np.uint8, np.uint16, np.uint32, np.uint64)
289
+ if np.iinfo(dt).max >= hi
290
+ )
291
+ return table[indices].astype(dtype)
@@ -3,21 +3,29 @@ from pathlib import Path
3
3
  import numpy as np
4
4
  import numpy.typing as npt
5
5
 
6
+ from . import pack
6
7
  from .image import MetaImage
7
8
 
8
9
 
9
10
  def load(path: Path | str) -> MetaImage:
10
11
  npz = np.load(path, allow_pickle=False)
11
- arr_float = _dequantize(npz["data"], npz["min"], npz["max"])
12
+ if "data" in npz:
13
+ arr = _dequantize(npz["data"], npz["min"], npz["max"])
14
+ else:
15
+ arr = (
16
+ pack.unpack_compact(npz["packed"])
17
+ .reshape(npz["packed_shape"])
18
+ .astype(np.dtype(str(npz["packed_dtype"])))
19
+ )
12
20
  mask = npz.get("mask")
13
21
  if mask is not None:
14
22
  shape = npz["mask_shape"]
15
23
  mask = np.unpackbits(mask, count=np.prod(shape)).reshape(shape).astype(bool)
16
- data = np.full_like(mask, np.nan, dtype=np.float32)
17
- data[mask] = arr_float
18
- arr_float = data
24
+ data = np.full_like(mask, np.nan, dtype=arr.dtype)
25
+ data[mask] = arr
26
+ arr = data
19
27
  return MetaImage(
20
- arr_float,
28
+ arr,
21
29
  spacing=npz["spacing"],
22
30
  direction=npz["direction"],
23
31
  origin=npz["origin"],
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tiny-metaio
3
- Version: 0.3.0
3
+ Version: 0.4.0
4
4
  Summary: Read and write MetaImages with minimal dependencies
5
5
  Author-email: Nicolas Cedilnik <nicolas.cedilnik@inria.fr>
6
6
  Project-URL: Homepage, https://gitlab.inria.fr/ncedilni/metaio
@@ -26,7 +26,7 @@ images (`.nii` or `.nii.gz`)
26
26
  and write `.mha` images
27
27
  in python with minimal dependencies.
28
28
 
29
- Bonus: provides a custom 8-bit MHA-like format when size matters: `.metaq`.
29
+ Bonus: provides a custom MHA-like format when size matters: `.metaq`.
30
30
 
31
31
  ## Installation
32
32
 
@@ -61,6 +61,27 @@ array([0., 0.])
61
61
 
62
62
  ### Custom format
63
63
 
64
+ #### Integers
65
+
66
+ Integers are stored without loss in a very compact way.
67
+
68
+ ```python
69
+ >>> from pathlib import Path
70
+ >>> import numpy as np
71
+ >>> img = metaio.MetaImage([[-5] * 1000, [2] * 1000])
72
+ >>> img.save("some-name.mha")
73
+ >>> Path("some-name.mha").stat().st_size
74
+ 16246
75
+ >>> img.save_compact("some-name.metaq")
76
+ >>> round(Path("some-name.metaq").stat().st_size / Path("some-name.mha").stat().st_size, 2)
77
+ 0.09
78
+ >>> np.all(metaio.read("some-name.metaq").data == img.data)
79
+ np.True_
80
+
81
+ ```
82
+
83
+ #### Floats
84
+
64
85
  You need to specify a range of values covered by the quantization.
65
86
  Values outside this ranges will be clipped.
66
87
  A loss of precision is expected.
@@ -68,7 +89,7 @@ NaNs will be implicitly converted to zero.
68
89
 
69
90
  ```python
70
91
  >>> img = metaio.MetaImage([[-0.5, 1], [2.5, 5]])
71
- >>> img.save_quantized("some-name.metaq", min_=0, max_=2.5)
92
+ >>> img.save_compact("some-name.metaq", min_=0, max_=2.5)
72
93
  >>> metaio.read("some-name.metaq").data
73
94
  array([[0. , 1. ],
74
95
  [2.5, 2.5]], dtype=float32)
@@ -81,7 +102,7 @@ Values where the mask is `False` will not be stored at all and restored as NaN o
81
102
 
82
103
  ```python
83
104
  >>> img = metaio.MetaImage([[-0.5, 1], [2.5, 5]])
84
- >>> img.save_quantized("some-name.metaq", min_=0, max_=2.5, mask=img.data <= 2.5)
105
+ >>> img.save_compact("some-name.metaq", min_=0, max_=2.5, mask=img.data <= 2.5)
85
106
  >>> metaio.read("some-name.metaq").data
86
107
  array([[0. , 1. ],
87
108
  [2.5, nan]], dtype=float32)
@@ -10,6 +10,7 @@ src/metaio/__main__.py
10
10
  src/metaio/image.py
11
11
  src/metaio/mha.py
12
12
  src/metaio/nifti.py
13
+ src/metaio/pack.py
13
14
  src/metaio/py.typed
14
15
  src/metaio/quant.py
15
16
  src/metaio/write.py
@@ -18,6 +19,8 @@ src/tiny_metaio.egg-info/SOURCES.txt
18
19
  src/tiny_metaio.egg-info/dependency_links.txt
19
20
  src/tiny_metaio.egg-info/entry_points.txt
20
21
  src/tiny_metaio.egg-info/requires.txt
22
+ src/tiny_metaio.egg-info/scm_file_list.json
23
+ src/tiny_metaio.egg-info/scm_version.json
21
24
  src/tiny_metaio.egg-info/top_level.txt
22
25
  tests/conftest.py
23
26
  tests/test_quantization.py
@@ -0,0 +1,24 @@
1
+ {
2
+ "files": [
3
+ ".gitlab-ci.yml",
4
+ "pyproject.toml",
5
+ "Containerfile",
6
+ "uv.lock",
7
+ "README.md",
8
+ ".gitignore",
9
+ ".pre-commit-config.yaml",
10
+ "src/metaio/nifti.py",
11
+ "src/metaio/mha.py",
12
+ "src/metaio/__init__.py",
13
+ "src/metaio/__main__.py",
14
+ "src/metaio/py.typed",
15
+ "src/metaio/image.py",
16
+ "src/metaio/quant.py",
17
+ "src/metaio/pack.py",
18
+ "src/metaio/write.py",
19
+ "tests/test_symmetry.py",
20
+ "tests/test_vs_simpleitk.py",
21
+ "tests/conftest.py",
22
+ "tests/test_quantization.py"
23
+ ]
24
+ }
@@ -0,0 +1,8 @@
1
+ {
2
+ "tag": "0.4.0",
3
+ "distance": 0,
4
+ "node": "g6400fde85c37e549dd6571a725a32656bbed64b2",
5
+ "dirty": false,
6
+ "branch": "HEAD",
7
+ "node_date": "2026-07-23"
8
+ }
@@ -3,14 +3,12 @@ from pathlib import Path
3
3
  import numpy as np
4
4
  import pytest
5
5
 
6
- from metaio import MetaImage, read
6
+ from metaio import MetaImage, pack, read
7
7
  from metaio.image import _dict_to_structured_array, _quantize
8
8
  from metaio.quant import _dequantize, _structured_array_to_dict
9
9
 
10
10
 
11
- @pytest.mark.parametrize(
12
- "dtype", [np.float16, np.float32, np.int64, np.int32, np.uint16, np.uint32]
13
- )
11
+ @pytest.mark.parametrize("dtype", [np.float16, np.float32, np.float64, np.float128])
14
12
  @pytest.mark.parametrize("shape", [(10, 11), (3, 4, 5)])
15
13
  @pytest.mark.parametrize("min_,max_", [(0, 35), (-10, 700), (-150, 800)])
16
14
  def test_symmetry(
@@ -27,7 +25,7 @@ def test_symmetry(
27
25
  direction=np.eye(len(shape)).ravel(),
28
26
  metadata=meta,
29
27
  )
30
- img1.save_quantized(path, min_, max_)
28
+ img1.save_compact(path, min_, max_)
31
29
 
32
30
  below_mask = arr1 <= min_
33
31
  above_mask = arr1 >= max_
@@ -54,6 +52,58 @@ def test_symmetry(
54
52
  assert img2.metadata[k] == v
55
53
 
56
54
 
55
+ @pytest.mark.parametrize(
56
+ "dtype",
57
+ [
58
+ # np.uint8,
59
+ # np.uint16,
60
+ # np.uint32,
61
+ # np.uint64,
62
+ np.int8,
63
+ np.int16,
64
+ np.int64,
65
+ ],
66
+ )
67
+ @pytest.mark.parametrize("shape", [(10, 11), (3, 4, 5)])
68
+ @pytest.mark.parametrize("random_range", [10, 20, 1000, 10000, 100000, 1000000])
69
+ def test_pack(
70
+ tmp_path: Path, dtype: np.dtype, shape: tuple[int, ...], random_range: int
71
+ ) -> None:
72
+ path = tmp_path / "image.metaq"
73
+ meta = {"k1": "v1", "k2": "v2", "k3": "v3"}
74
+
75
+ arr1 = np.random.randint(
76
+ 0,
77
+ random_range if np.issubdtype(dtype, np.unsignedinteger) else random_range // 2,
78
+ size=shape,
79
+ ).astype(dtype)
80
+
81
+ if np.issubdtype(dtype, np.signedinteger):
82
+ arr1 -= min(np.iinfo(dtype).max, random_range // 2)
83
+
84
+ img1 = MetaImage(
85
+ arr1,
86
+ spacing=np.array([0.5] * len(shape)),
87
+ origin=np.array([-4] * len(shape)),
88
+ direction=np.eye(len(shape)).ravel(),
89
+ metadata=meta,
90
+ )
91
+ img1.save_compact(path)
92
+
93
+ img2 = read(path)
94
+ arr2 = img2.data
95
+
96
+ assert np.all(arr1 == arr2)
97
+
98
+ assert arr1.dtype == arr2.dtype
99
+ assert np.all(img1.spacing == img2.spacing)
100
+ assert np.all(img1.origin == img2.origin)
101
+ assert np.all(img1.direction == img2.direction)
102
+
103
+ for k, v in meta.items():
104
+ assert img2.metadata[k] == v
105
+
106
+
57
107
  def test_quantize() -> None:
58
108
  assert np.all(_quantize(np.array([1, 2, 3, 4, 5]), 2, 4) == [0, 0, 128, 255, 255])
59
109
  assert np.all(_quantize(np.array([-4, 0, 4]), -4, 4) == [0, 128, 255])
@@ -77,3 +127,9 @@ def test_dict_to_struct() -> None:
77
127
  arr = _dict_to_structured_array(dct1)
78
128
  dct2 = _structured_array_to_dict(arr)
79
129
  assert dct1 == dct2
130
+
131
+
132
+ def test_pack_unpack() -> None:
133
+ assert np.all(
134
+ pack.unpack_compact(pack.pack_compact([-3, 0, 1, -3, 1])) == [-3, 0, 1, -3, 1]
135
+ )
File without changes
File without changes
File without changes
File without changes
File without changes