euler-loading 2.25.0__tar.gz → 2.28.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. {euler_loading-2.25.0 → euler_loading-2.28.0}/.gitignore +5 -0
  2. euler_loading-2.28.0/CHANGELOG.md +58 -0
  3. {euler_loading-2.25.0 → euler_loading-2.28.0}/PKG-INFO +32 -1
  4. {euler_loading-2.25.0 → euler_loading-2.28.0}/README.md +31 -0
  5. {euler_loading-2.25.0 → euler_loading-2.28.0}/docs/README.md +1 -1
  6. {euler_loading-2.25.0 → euler_loading-2.28.0}/docs/loaders.md +150 -4
  7. euler_loading-2.28.0/euler_loading/__main__.py +10 -0
  8. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/_ds_crawler_utils.py +18 -0
  9. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/_resolution.py +33 -8
  10. euler_loading-2.28.0/euler_loading/cli.py +320 -0
  11. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/dataset.py +6 -6
  12. euler_loading-2.28.0/euler_loading/dry_run.py +723 -0
  13. euler_loading-2.28.0/euler_loading/loaders/cpu/synscapes.py +769 -0
  14. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/generate/loaders.json +26 -0
  15. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/gpu/synscapes.py +64 -6
  16. {euler_loading-2.25.0 → euler_loading-2.28.0}/pyproject.toml +4 -1
  17. euler_loading-2.28.0/tests/test_dry_run.py +592 -0
  18. euler_loading-2.28.0/tests/test_synscapes.py +949 -0
  19. euler_loading-2.25.0/CHANGELOG.md +0 -25
  20. euler_loading-2.25.0/euler_loading/loaders/cpu/synscapes.py +0 -205
  21. euler_loading-2.25.0/tests/test_synscapes.py +0 -319
  22. {euler_loading-2.25.0 → euler_loading-2.28.0}/LICENSE +0 -0
  23. {euler_loading-2.25.0 → euler_loading-2.28.0}/docs/dataset.md +0 -0
  24. {euler_loading-2.25.0 → euler_loading-2.28.0}/docs/materialization.md +0 -0
  25. {euler_loading-2.25.0 → euler_loading-2.28.0}/docs/preprocessing.md +0 -0
  26. {euler_loading-2.25.0 → euler_loading-2.28.0}/docs/transform-descriptors.md +0 -0
  27. {euler_loading-2.25.0 → euler_loading-2.28.0}/docs/writing.md +0 -0
  28. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/__init__.py +0 -0
  29. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/_dataset_contract.py +0 -0
  30. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/_metadata.py +0 -0
  31. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/_writing.py +0 -0
  32. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/geometry.py +0 -0
  33. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/indexing.py +0 -0
  34. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/__init__.py +0 -0
  35. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/_annotations.py +0 -0
  36. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/_princeton_dense.py +0 -0
  37. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/_writer_utils.py +0 -0
  38. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/contracts.py +0 -0
  39. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/cpu/__init__.py +0 -0
  40. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/cpu/generic.py +0 -0
  41. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/cpu/generic_dense_depth.py +0 -0
  42. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/cpu/muses.py +0 -0
  43. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/cpu/princeton_dense.py +0 -0
  44. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/cpu/real_drive_sim.py +0 -0
  45. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/cpu/vkitti2.py +0 -0
  46. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/generate/__init__.py +0 -0
  47. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/generate/__main__.py +0 -0
  48. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/generic.py +0 -0
  49. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/gpu/__init__.py +0 -0
  50. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/gpu/generic.py +0 -0
  51. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/gpu/generic_dense_depth.py +0 -0
  52. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/gpu/muses.py +0 -0
  53. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/gpu/princeton_dense.py +0 -0
  54. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/gpu/real_drive_sim.py +0 -0
  55. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/gpu/vkitti2.py +0 -0
  56. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/materialized.py +0 -0
  57. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/muses.py +0 -0
  58. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/princeton_dense.py +0 -0
  59. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/real_drive_sim.py +0 -0
  60. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/synscapes.py +0 -0
  61. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/loaders/vkitti2.py +0 -0
  62. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/materialization.py +0 -0
  63. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/output_encoding.py +0 -0
  64. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/preprocessing.py +0 -0
  65. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/receipts.py +0 -0
  66. {euler_loading-2.25.0 → euler_loading-2.28.0}/euler_loading/transform_descriptors.py +0 -0
  67. {euler_loading-2.25.0 → euler_loading-2.28.0}/examples/README.md +0 -0
  68. {euler_loading-2.25.0 → euler_loading-2.28.0}/examples/real_drive_sim_preview.py +0 -0
  69. {euler_loading-2.25.0 → euler_loading-2.28.0}/examples/vkitti2_sample.py +0 -0
  70. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/__init__.py +0 -0
  71. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/conftest.py +0 -0
  72. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/example_rds_calib.json +0 -0
  73. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/test_dataset.py +0 -0
  74. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/test_id_schema.py +0 -0
  75. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/test_indexing.py +0 -0
  76. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/test_loaders.py +0 -0
  77. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/test_materialization.py +0 -0
  78. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/test_preprocessing.py +0 -0
  79. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/test_python_compat.py +0 -0
  80. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/test_real_dataset.py +0 -0
  81. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/test_transform_descriptors.py +0 -0
  82. {euler_loading-2.25.0 → euler_loading-2.28.0}/tests/test_writing.py +0 -0
@@ -42,3 +42,8 @@ TASK.md
42
42
 
43
43
  # uv
44
44
  uv.lock
45
+
46
+ # Stray npm lockfile. Nothing in this package is built with npm, and worktree
47
+ # tooling rewrites its "name" field to the branch directory on every branch,
48
+ # so tracking it produced a merge conflict on every quick-merge.
49
+ package-lock.json
@@ -0,0 +1,58 @@
1
+ # Changelog
2
+
3
+ ## 2.27.0 (unreleased)
4
+
5
+ - Add a dry-run command: `euler-loading <path>` (also `python -m
6
+ euler_loading <path>`) walks the ds-crawler artifact sets at or below one
7
+ folder, resolves the loader each dataset head declares in
8
+ `addons.euler_loading`, decodes a sample of the indexed files with it, and
9
+ reports shapes, dtypes and value ranges. Directories and `.zip` archives
10
+ are treated alike, nothing is written, and the exit status is non-zero when
11
+ a modality fails to resolve or decode. `euler_loading.dry_run.dry_run()`
12
+ exposes the same report to Python.
13
+ - `resolve_loader_module` and `resolve_writer_module` take a
14
+ `variant="gpu"|"cpu"` keyword, so the CPU modules can be resolved through
15
+ the same contract pathway as the torch ones. This is what the dry run's
16
+ `--cpu` uses to check archives on a torch-free install.
17
+
18
+ ## 2.26.0
19
+
20
+ - Add `read_extrinsics` to the Synscapes loaders, building a 4x4 rigid
21
+ transform from the six `camera.extrinsic` scalars. `transform_direction`
22
+ and `camera_axes` select the pose, its inverse, and vehicle or optical
23
+ camera axes; the assumed `Rz(yaw) @ Ry(pitch) @ Rx(roll)` order is recorded
24
+ in the modality metadata because the dataset does not document it.
25
+ - Add Synscapes writers for every modality, making the loader a
26
+ `DenseDepthCodec`. Depth writes a float32 EXR `Z` channel and needs a
27
+ filesystem path; the intrinsics and extrinsics writers merge into a single
28
+ `meta/<id>.json`, replacing it in one step and keeping its permissions,
29
+ symlink and unmodelled fields rather than truncating it in place.
30
+ - `sky_mask` now honours `meta['sky_class_id']`, or the same key in per-file
31
+ attributes, instead of hardcoding Cityscapes ID 23, so a dataset that
32
+ relabels sky round-trips through `write_sky_mask`.
33
+ - Accept the plural `camera.intrinsics` / `camera.extrinsics` spellings and
34
+ report missing camera fields by name instead of raising `KeyError`.
35
+
36
+ ## 2.24.0 (unreleased)
37
+
38
+ Add source-backed capture, strict per-output writers, explicit NPY/PNG encoding, typed calibration views and replay. Consolidate pinhole geometry and correct legacy skew scaling.
39
+
40
+
41
+ ## 2.23.0 (source changes; not published)
42
+
43
+ - Reject wrapped decoders during dataset export by checking actual built-in
44
+ function identity. Ordinary callable loading remains supported.
45
+ - Apply boolean thresholds before Pillow dtype conversion, and permit exact
46
+ int32/int64 crop and identity-resize plans while refusing float32 resampling.
47
+ - Add `SamplePreprocessor.export_descriptor`, `MultiModalDataset.export_transform_plan`,
48
+ `SerializableTransform`, and `resolve_transform_descriptor`, paired with contract 0.4.0.
49
+ - Freeze profiles, qualified source/calibration bindings, field policies, operation
50
+ order, and installed CPU backend versions. Infer output profiles and virtual
51
+ calibration geometry without materialization claims.
52
+ - Add explicit Torch/Pillow execution, corrected pinhole skew in the opt-in path,
53
+ validity-aware depth policy, source immutability, and worker validator registration.
54
+ - Cover five-field round trips, projected geometry, ambiguous bindings, unsupported
55
+ semantics, backend choices, schemas, and spawned workers in synthetic tests.
56
+
57
+ Legacy preprocessing and writer behavior remain unchanged. Execution receipts,
58
+ materialized metadata propagation, producer wiring, and GT replay are later work.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: euler-loading
3
- Version: 2.25.0
3
+ Version: 2.28.0
4
4
  Summary: Multi-modal PyTorch dataloader using ds-crawler indices
5
5
  Project-URL: Homepage, https://github.com/d-rothen/euler-loading
6
6
  Project-URL: Repository, https://github.com/d-rothen/euler-loading
@@ -174,6 +174,36 @@ when you want the CPU variant or a custom callable. See
174
174
  [Automatic loader resolution](docs/loaders.md#automatic-loader-resolution) for
175
175
  the full contract and writer rules.
176
176
 
177
+ ## Checking a dataset
178
+
179
+ Before any of that runs, point the bundled command at a folder to see what
180
+ euler-loading would make of it:
181
+
182
+ ```bash
183
+ euler-loading /data/vkitti2
184
+ ```
185
+
186
+ For every ds-crawler artifact set below that path it resolves the loader the
187
+ dataset declares, decodes a sample of the indexed files with it, and prints
188
+ the shape, dtype and value range that came back:
189
+
190
+ ```text
191
+ vkitti_2.0.3_rgb [ok]
192
+ path /data/vkitti2/vkitti_2.0.3_rgb
193
+ contract vkitti2_rgb, "Virtual KITTI 2 RGB", modality key: rgb
194
+ loader vkitti2.rgb -> euler_loading.loaders.gpu.vkitti2.rgb
195
+ index 21260 files; available splits: train, val
196
+ decoded 1/1
197
+ Scene01/clone/frames/rgb/Camera_0/rgb_00000.jpg -> Tensor (3, 375, 1242) float32 in [0, 1] (11.4 ms)
198
+ ```
199
+
200
+ Nothing is written, directories and `.zip` archives are treated alike, and the
201
+ exit status is non-zero if any modality failed to resolve or decode — so it
202
+ also works as a CI step. `--all` decodes every file instead of a sample,
203
+ `--cpu` checks the NumPy loaders on a torch-free install, and `--json` prints
204
+ the same report for a machine. See
205
+ [Dry-running loaders](docs/loaders.md#dry-running-loaders).
206
+
177
207
  ## What you get
178
208
 
179
209
  | | |
@@ -186,6 +216,7 @@ the full contract and writer rules.
186
216
  | **Loader resolution** | Loaders and writers resolve from the `dataset-head.json` `addons.euler_loading` contract, so datasets describe how to read themselves. |
187
217
  | **Writing back** | Resolved writers put inference outputs back in dataset-native formats, re-indexable with matching IDs. |
188
218
  | **Spatial preprocessing** | `SamplePreprocessor` resizes and crops consistently across images, depth, masks, ray maps *and* intrinsics. |
219
+ | **Dry runs** | `euler-loading <folder>` resolves and exercises every loader a dataset declares, without writing anything. |
189
220
 
190
221
  ## Built-in loaders
191
222
 
@@ -135,6 +135,36 @@ when you want the CPU variant or a custom callable. See
135
135
  [Automatic loader resolution](docs/loaders.md#automatic-loader-resolution) for
136
136
  the full contract and writer rules.
137
137
 
138
+ ## Checking a dataset
139
+
140
+ Before any of that runs, point the bundled command at a folder to see what
141
+ euler-loading would make of it:
142
+
143
+ ```bash
144
+ euler-loading /data/vkitti2
145
+ ```
146
+
147
+ For every ds-crawler artifact set below that path it resolves the loader the
148
+ dataset declares, decodes a sample of the indexed files with it, and prints
149
+ the shape, dtype and value range that came back:
150
+
151
+ ```text
152
+ vkitti_2.0.3_rgb [ok]
153
+ path /data/vkitti2/vkitti_2.0.3_rgb
154
+ contract vkitti2_rgb, "Virtual KITTI 2 RGB", modality key: rgb
155
+ loader vkitti2.rgb -> euler_loading.loaders.gpu.vkitti2.rgb
156
+ index 21260 files; available splits: train, val
157
+ decoded 1/1
158
+ Scene01/clone/frames/rgb/Camera_0/rgb_00000.jpg -> Tensor (3, 375, 1242) float32 in [0, 1] (11.4 ms)
159
+ ```
160
+
161
+ Nothing is written, directories and `.zip` archives are treated alike, and the
162
+ exit status is non-zero if any modality failed to resolve or decode — so it
163
+ also works as a CI step. `--all` decodes every file instead of a sample,
164
+ `--cpu` checks the NumPy loaders on a torch-free install, and `--json` prints
165
+ the same report for a machine. See
166
+ [Dry-running loaders](docs/loaders.md#dry-running-loaders).
167
+
138
168
  ## What you get
139
169
 
140
170
  | | |
@@ -147,6 +177,7 @@ the full contract and writer rules.
147
177
  | **Loader resolution** | Loaders and writers resolve from the `dataset-head.json` `addons.euler_loading` contract, so datasets describe how to read themselves. |
148
178
  | **Writing back** | Resolved writers put inference outputs back in dataset-native formats, re-indexable with matching IDs. |
149
179
  | **Spatial preprocessing** | `SamplePreprocessor` resizes and crops consistently across images, depth, masks, ray maps *and* intrinsics. |
180
+ | **Dry runs** | `euler-loading <folder>` resolves and exercises every loader a dataset declares, without writing anything. |
150
181
 
151
182
  ## Built-in loaders
152
183
 
@@ -3,7 +3,7 @@
3
3
  | Guide | Covers |
4
4
  |---|---|
5
5
  | [Dataset & modalities](dataset.md) | `Modality` and `MultiModalDataset` reference, the sample dict, hierarchical modalities, splits, scoped metadata, zip archives, layout-aware loading |
6
- | [Loaders & writers](loaders.md) | The loader contract, per-file attributes, automatic resolution, protocols, and the full built-in loader inventory |
6
+ | [Loaders & writers](loaders.md) | The loader contract, per-file attributes, automatic resolution, the `euler-loading` dry run, protocols, and the full built-in loader inventory |
7
7
  | [Preprocessing & transforms](preprocessing.md) | Cross-modal transforms, `SamplePreprocessor`, field kinds, calibration-aware resize and crop |
8
8
  | [Writing outputs](writing.md) | Writing predictions back in dataset-native formats and re-indexing them |
9
9
 
@@ -7,6 +7,7 @@ that file means.
7
7
  - [The contract](#the-contract)
8
8
  - [Per-file attributes](#per-file-attributes)
9
9
  - [Automatic loader resolution](#automatic-loader-resolution)
10
+ - [Dry-running loaders](#dry-running-loaders)
10
11
  - [Loader protocols](#loader-protocols)
11
12
  - [Built-in loaders](#built-in-loaders)
12
13
 
@@ -129,6 +130,102 @@ Writers resolve from the same `addons.euler_loading` entry, in this order:
129
130
  2. for `function: "read_<suffix>"`, `write_<suffix>`,
130
131
  3. `write_<function>`.
131
132
 
133
+ ## Dry-running loaders
134
+
135
+ Resolution only pays off if the loader a dataset names can really read its
136
+ files. Point the bundled command at a path to find out, before a training run
137
+ depends on it:
138
+
139
+ ```bash
140
+ euler-loading /data/vkitti2
141
+ python -m euler_loading /data/vkitti2 # same thing, no console script needed
142
+ ```
143
+
144
+ It takes one path — a modality root, a `.zip` archive, or a folder holding
145
+ several of them — and for every ds-crawler artifact set below it:
146
+
147
+ 1. loads the index, or a named split,
148
+ 2. resolves the loader its `addons.euler_loading` entry declares,
149
+ 3. decodes a sample of the indexed files with that loader,
150
+ 4. reports the shape, dtype and value range of what came back.
151
+
152
+ Nothing is written and no dataset is constructed. A modality that cannot be
153
+ read is reported, not raised, so one broken archive does not hide the rest.
154
+
155
+ ```text
156
+ euler-loading dry-run: /data/vkitti2
157
+ loader variant: gpu (torch tensors)
158
+
159
+ vkitti_2.0.3_rgb [ok]
160
+ path /data/vkitti2/vkitti_2.0.3_rgb
161
+ contract vkitti2_rgb, "Virtual KITTI 2 RGB", modality key: rgb
162
+ loader vkitti2.rgb -> euler_loading.loaders.gpu.vkitti2.rgb
163
+ writer write_rgb
164
+ index 21260 files; available splits: train, val
165
+ decoded 1/1
166
+ Scene01/clone/frames/rgb/Camera_0/rgb_00000.jpg -> Tensor (3, 375, 1242) float32 in [0, 1] (11.4 ms)
167
+
168
+ vkitti_2.0.3_textgt [ok]
169
+ path /data/vkitti2/vkitti_2.0.3_textgt
170
+ contract vkitti2_intrinsics, "Virtual KITTI 2 intrinsics", modality key: camera_intrinsics
171
+ loader vkitti2.read_intrinsics -> euler_loading.loaders.gpu.vkitti2.read_intrinsics (hierarchical)
172
+ writer write_intrinsics
173
+ index 50 files
174
+ decoded 1/1
175
+ Scene01/clone/intrinsic.txt -> Tensor (3, 3) float32 in [0, 725] (0.6 ms)
176
+
177
+ 2 modalities checked: 2 ok, 0 failed; 2/2 files decoded
178
+ ```
179
+
180
+ The exit status is 0 when every modality resolved a loader and decoded every
181
+ file it was asked to, and 1 otherwise, so the command works as a CI step or a
182
+ job prologue.
183
+
184
+ | Flag | Effect |
185
+ |---|---|
186
+ | `-n, --samples N` | Files to decode per modality, spread evenly across the index (default 1). `0` resolves loaders without decoding anything. |
187
+ | `--all` | Decode every indexed file. Slow, but checks the whole archive. |
188
+ | `--split NAME` | Dry-run a named ds-crawler split instead of the canonical index. |
189
+ | `--scope SCOPE` | Read `.ds_crawler/SCOPE/` only, instead of every artifact set a root carries. |
190
+ | `--cpu` | Resolve the NumPy loaders instead of the torch ones. Works on an install without PyTorch. |
191
+ | `--max-depth N` | Directory levels below the path to search for dataset roots (default 3). |
192
+ | `--json` | Print the report as JSON instead of text. |
193
+ | `-v, --verbose` | Log debug output, including a traceback for every failed load. |
194
+
195
+ The path takes the same inline selectors as `Modality`, so
196
+ `euler-loading /data/muses.zip:train#scope=rgb` checks one split of one scope.
197
+
198
+ Discovery stops at the first dataset root down each branch: a directory
199
+ carrying its own `.ds_crawler/` is checked, never descended into, so a
200
+ dataset's own scene directories are never walked. A root with several
201
+ metadata scopes contributes one entry per scope, labelled with the selector
202
+ that addresses it.
203
+
204
+ ### Reading the report
205
+
206
+ | Line | What it is telling you |
207
+ |---|---|
208
+ | `error loader: ... no 'addons.euler_loading' entry` | the dataset head declares no loader, so only an explicit `Modality(..., loader=...)` can read it |
209
+ | `loader vkitti2.rgb (unresolved)` plus an `error` | the contract names a loader that is unknown, misspelled, or needs a dependency this install lacks — `--cpu` resolves the torch-free variant |
210
+ | `decoded 0/2` with per-file `FAILED` | the files are indexed and present, but the declared loader cannot decode them |
211
+ | `(hierarchical)` after the loader | the declared function reads per-scene data; pass that modality as `hierarchical_modalities`, not `modalities` |
212
+ | `error index: ...` | ds-crawler could not produce an index — a missing split, an unreadable archive, or a dataset head the contract rejects |
213
+ | a warning about shared file IDs | the modalities below the path have no file ID in common, so they cannot be intersected into one `MultiModalDataset` |
214
+
215
+ The same report is available as an API, which is what the command is a thin
216
+ shell around:
217
+
218
+ ```python
219
+ from euler_loading.dry_run import dry_run
220
+
221
+ report = dry_run("/data/vkitti2", samples=3)
222
+ if not report.ok:
223
+ for modality in report.modalities:
224
+ print(modality.label, modality.errors)
225
+
226
+ report.to_dict() # the same structure --json prints
227
+ ```
228
+
132
229
  ## Loader protocols
133
230
 
134
231
  `DenseDepthLoader` is a `runtime_checkable` Protocol defining the loader
@@ -193,8 +290,26 @@ Writers exist for every modality above.
193
290
  | `depth` | 1HW / HW | float32 | `img/depth/<id>.exr`, `Z` channel, planar depth in metres |
194
291
  | `class_segmentation` | 1HW / HW | int64 | `img/class/<id>.png`, original Cityscapes label IDs |
195
292
  | `instance_segmentation` | 1HW / HW | int64 | `img/instance/<id>.png`, decoded as `R + 256 * G + 65536 * B` |
196
- | `sky_mask` | 1HW / HW | bool | Sky label ID `23` in `img/class/<id>.png` |
293
+ | `sky_mask` | 1HW / HW | bool | Sky label in `img/class/<id>.png`; ID `23` by default, or `meta['sky_class_id']` |
197
294
  | `read_intrinsics` | 3×3 | float32 | `camera.intrinsic` from `meta/<id>.json`: `fx`, `fy`, `u0`, `v0` |
295
+ | `read_extrinsics` | 4×4 | float32 | `camera.extrinsic` from `meta/<id>.json`: `x`, `y`, `z`, `pitch`, `roll`, `yaw` |
296
+
297
+ Writers exist for every modality above. `write_depth` needs a filesystem path
298
+ because OpenEXR cannot write to a stream; dataset writers hand it a temporary
299
+ file automatically, so zip outputs work unchanged. `write_sky_mask` emits a
300
+ class image labelling sky with `meta['sky_class_id']`, default 23, and
301
+ `write_intrinsics` and `write_extrinsics` merge into one `meta/<id>.json`
302
+ rather than overwriting each other or the file's `scene` and `instance`
303
+ blocks.
304
+
305
+ That merge reads the file back, so it applies to filesystem destinations and
306
+ replaces the file atomically. A zip destination is handed a fresh stream per
307
+ modality instead, so writing both camera modalities into the **same** archive
308
+ produces two entries with the same name, of which only the last is
309
+ retrievable — give each modality its own output writer. Because the merge
310
+ keeps fields it does not model, writing a rescaled matrix without passing
311
+ `meta['resx']`/`['resy']` leaves the previous resolution in place; supply them
312
+ whenever the matrix no longer matches the recorded resolution.
198
313
 
199
314
  Formats follow the [Synscapes dataset reference](https://synscapes.on.liu.se/features.html)
200
315
  and [FoggySynscapes EXR reader](https://github.com/MartinHahner/FoggySynscapes/blob/main/source/Depth_processing/exr_to_mat.py).
@@ -202,14 +317,14 @@ Depth values are already in metres, so they are returned unchanged, including
202
317
  non-finite values. Semantic labels retain the original IDs, including void
203
318
  labels, without remapping to training IDs.
204
319
 
205
- EXR loading needs the optional OpenEXR dependency:
320
+ EXR depth loading and writing need the optional OpenEXR dependency:
206
321
 
207
322
  ```bash
208
323
  pip install "euler-loading[gpu,synscapes]" # omit gpu for NumPy-only use
209
324
  ```
210
325
 
211
- RGB, segmentation and intrinsics work without OpenEXR. All six functions
212
- accept paths and binary streams. Intrinsics are stored in per-image files,
326
+ RGB, segmentation, intrinsics and extrinsics work without OpenEXR; only depth
327
+ loading and writing need it. All readers accept paths and binary streams. Intrinsics are stored in per-image files,
213
328
  so the native `meta` directory is a regular modality:
214
329
 
215
330
  ```python
@@ -224,6 +339,37 @@ dataset = MultiModalDataset(modalities={
224
339
  })
225
340
  ```
226
341
 
342
+ #### Extrinsics conventions
343
+
344
+ Synscapes records the camera mount as six scalars — `x`, `y`, `z` in metres and
345
+ `pitch`, `roll`, `yaw` in radians — in the ego-vehicle frame, which the dataset
346
+ defines as x forward, y left, z up. `read_extrinsics` composes them into a
347
+ rigid 4×4 matrix, by default the camera's pose on the vehicle
348
+ (`X_ego = T @ X_camera`). Two keys, taken from per-file `attributes` first and
349
+ then from `meta`, select the other useful conventions:
350
+
351
+ | Key | Values | Meaning |
352
+ |---|---|---|
353
+ | `transform_direction` | `camera_to_ego` (default), `ego_to_camera` | Direction of the returned transform |
354
+ | `camera_axes` | `vehicle` (default), `optical` | Camera axes: vehicle-aligned, or x right / y down / z forward |
355
+
356
+ Combine `ego_to_camera` with `optical` to project ego-frame points — such as
357
+ the 3D bounding boxes in the instance metadata — through `read_intrinsics`:
358
+
359
+ ```python
360
+ T = synscapes.read_extrinsics(
361
+ "/data/Synscapes/meta/1.json",
362
+ {"transform_direction": "ego_to_camera", "camera_axes": "optical"},
363
+ )
364
+ ```
365
+
366
+ The dataset documents the six fields and the ego frame but not the order in
367
+ which the angles compose, so the loaders apply the usual automotive
368
+ `Rz(yaw) @ Ry(pitch) @ Rx(roll)` and record that in `loaders.json` under the
369
+ modality's `rotation_order`. The values are constant across the released
370
+ dataset. `write_extrinsics` reads the same two keys, so a matrix loaded with
371
+ one set of options writes back unchanged with the same options.
372
+
227
373
  Automatic resolution uses `loader="synscapes"` in the dataset contract.
228
374
  Intrinsics describe the metadata's image resolution (native 1440×720).
229
375
  When using `img/rgb-2k` at 2048×1024, scale the camera matrix with
@@ -0,0 +1,10 @@
1
+ """``python -m euler_loading`` — dry-run the loaders a dataset declares."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sys
6
+
7
+ from .cli import main
8
+
9
+ if __name__ == "__main__":
10
+ sys.exit(main())
@@ -1,5 +1,7 @@
1
1
  from __future__ import annotations
2
2
 
3
+ import io
4
+ import zipfile
3
5
  from collections.abc import Mapping
4
6
  from pathlib import Path
5
7
  from typing import Any
@@ -22,6 +24,22 @@ from ds_crawler.zip_utils import (
22
24
  )
23
25
 
24
26
 
27
+ def read_zip_member(
28
+ archive: zipfile.ZipFile,
29
+ prefix: str,
30
+ relative_path: str,
31
+ ) -> io.BytesIO:
32
+ """Read one member of a zip-backed modality into an in-memory buffer.
33
+
34
+ The buffer carries the modality-relative path as its ``name`` so loaders
35
+ that branch on the file extension behave the same as for a filesystem
36
+ path.
37
+ """
38
+ buffer = io.BytesIO(archive.read(prefix + relative_path))
39
+ buffer.name = relative_path
40
+ return buffer
41
+
42
+
25
43
  def as_non_empty_str(value: Any) -> str | None:
26
44
  if value is None:
27
45
  return None
@@ -18,6 +18,10 @@ if TYPE_CHECKING:
18
18
  logger = logging.getLogger(__name__)
19
19
 
20
20
 
21
+ #: Built-in loader variants. ``gpu`` returns torch tensors, ``cpu`` NumPy
22
+ #: arrays; automatic resolution from a dataset contract uses ``gpu``.
23
+ LOADER_VARIANTS = ("gpu", "cpu")
24
+
21
25
  _LOADER_MODULES: dict[str, str] = {
22
26
  "materialized": "euler_loading.loaders.materialized",
23
27
  "vkitti2": "euler_loading.loaders.gpu.vkitti2",
@@ -46,30 +50,45 @@ def _get_euler_loading_meta(index: Mapping[str, Any]) -> Mapping[str, Any] | Non
46
50
  return None
47
51
 
48
52
 
49
- def resolve_loader_module(name: str) -> ModuleType:
50
- """Import and return the GPU loader module for *name*.
53
+ def resolve_loader_module(name: str, *, variant: str = "gpu") -> ModuleType:
54
+ """Import and return the loader module for *name*.
51
55
 
52
56
  Example::
53
57
 
54
58
  module = resolve_loader_module("vkitti2")
55
59
  sky_fn = module.sky_mask # get a specific function
56
60
 
61
+ Args:
62
+ name: Loader name, as declared by ``addons.euler_loading.loader``.
63
+ variant: ``"gpu"`` for the torch loaders (the default, and what
64
+ automatic resolution uses) or ``"cpu"`` for the NumPy ones.
65
+
57
66
  Raises:
58
- ValueError: If *name* does not match any known loader.
67
+ ValueError: If *name* does not match any known loader, or *variant*
68
+ is neither ``"gpu"`` nor ``"cpu"``.
59
69
  """
70
+ if variant not in LOADER_VARIANTS:
71
+ available = ", ".join(LOADER_VARIANTS)
72
+ raise ValueError(
73
+ f"Unknown loader variant {variant!r}. Available variants: {available}"
74
+ )
60
75
  module_path = _LOADER_MODULES.get(name)
61
76
  if module_path is None:
62
77
  available = ", ".join(sorted(_LOADER_MODULES))
63
78
  raise ValueError(f"Unknown loader {name!r}. Available loaders: {available}")
79
+ if variant == "cpu":
80
+ # Loaders that exist in one variant only (``materialized``) are not
81
+ # registered under ``.gpu.`` and therefore stay as they are.
82
+ module_path = module_path.replace(".loaders.gpu.", ".loaders.cpu.")
64
83
  return importlib.import_module(module_path)
65
84
 
66
85
 
67
- def resolve_writer_module(name: str) -> ModuleType:
86
+ def resolve_writer_module(name: str, *, variant: str = "gpu") -> ModuleType:
68
87
  """Import and return the writer module for *name*.
69
88
 
70
89
  Writers live next to loader functions in the same modules.
71
90
  """
72
- return resolve_loader_module(name)
91
+ return resolve_loader_module(name, variant=variant)
73
92
 
74
93
 
75
94
  def _builtin_loader_id(loader: Callable[..., Any]) -> str | None:
@@ -119,8 +138,13 @@ def _resolve_loader(
119
138
  modality_name: str,
120
139
  modality: Modality,
121
140
  index: dict[str, Any],
141
+ variant: str = "gpu",
122
142
  ) -> Callable[..., Any]:
123
- """Return the effective loader for a modality."""
143
+ """Return the effective loader for a modality.
144
+
145
+ An explicit ``Modality.loader`` always wins; *variant* only selects
146
+ between the built-in GPU and CPU modules during automatic resolution.
147
+ """
124
148
  if modality.loader is not None:
125
149
  return modality.loader
126
150
 
@@ -143,7 +167,7 @@ def _resolve_loader(
143
167
  module_name: str = euler_loading_meta["loader"]
144
168
  func_name: str = euler_loading_meta["function"]
145
169
 
146
- module = resolve_loader_module(module_name)
170
+ module = resolve_loader_module(module_name, variant=variant)
147
171
 
148
172
  func = getattr(module, func_name, None)
149
173
  if func is None or not callable(func):
@@ -185,6 +209,7 @@ def _resolve_writer(
185
209
  modality_name: str,
186
210
  modality: Modality,
187
211
  index: dict[str, Any],
212
+ variant: str = "gpu",
188
213
  ) -> Callable[..., Any] | None:
189
214
  """Return the effective writer for a modality, if available."""
190
215
  if modality.writer is not None:
@@ -200,7 +225,7 @@ def _resolve_writer(
200
225
  return None
201
226
 
202
227
  try:
203
- module = resolve_writer_module(module_name)
228
+ module = resolve_writer_module(module_name, variant=variant)
204
229
  except ValueError:
205
230
  logger.warning(
206
231
  "Modality '%s': cannot resolve writer module %r.",