euler-eval 2.22.0__tar.gz → 2.26.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. euler_eval-2.26.0/LICENSE +21 -0
  2. euler_eval-2.26.0/PKG-INFO +237 -0
  3. euler_eval-2.26.0/README.md +188 -0
  4. euler_eval-2.26.0/euler_eval/THIRD_PARTY_NOTICES.md +83 -0
  5. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/__init__.py +4 -0
  6. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/cli.py +76 -36
  7. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/data.py +56 -18
  8. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/evaluate.py +94 -22
  9. euler_eval-2.26.0/euler_eval/metric_sets.py +109 -0
  10. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/__init__.py +69 -98
  11. euler_eval-2.26.0/euler_eval/metrics/_natural_scene_models.py +203 -0
  12. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/absrel.py +2 -1
  13. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/daniel_error.py +2 -1
  14. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/depth_binned_error.py +2 -1
  15. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/depth_edge_f1.py +2 -1
  16. euler_eval-2.26.0/euler_eval/metrics/fade.py +257 -0
  17. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/fid_kid.py +4 -4
  18. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/gpu_depth_batch.py +0 -1
  19. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/gpu_image_batch.py +5 -3
  20. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/high_freq_energy.py +1 -1
  21. euler_eval-2.26.0/euler_eval/metrics/niqe.py +247 -0
  22. euler_eval-2.26.0/euler_eval/metrics/normal_consistency.py +314 -0
  23. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/points3d_cloud.py +1 -3
  24. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/points3d_geometry.py +17 -9
  25. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/psnr.py +2 -1
  26. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/rgb_edge_f1.py +2 -1
  27. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/rgb_psnr_ssim.py +2 -1
  28. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/rmse.py +2 -1
  29. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/scale_invariant_log.py +2 -3
  30. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/ssim.py +2 -1
  31. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/tail_errors.py +2 -1
  32. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/utils.py +27 -81
  33. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/sanity_checker.py +3 -3
  34. euler_eval-2.26.0/euler_eval/utils/hierarchy_parser.py +54 -0
  35. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/validation.py +93 -3
  36. euler_eval-2.26.0/euler_eval.egg-info/PKG-INFO +237 -0
  37. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval.egg-info/SOURCES.txt +8 -0
  38. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval.egg-info/requires.txt +6 -4
  39. euler_eval-2.26.0/pyproject.toml +119 -0
  40. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_alignment.py +0 -1
  41. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_config.py +4 -31
  42. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_data.py +7 -33
  43. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_data_builders.py +43 -10
  44. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_depth_alignment_output.py +97 -32
  45. euler_eval-2.26.0/tests/test_domain_metrics.py +176 -0
  46. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_evaluate_helpers.py +0 -2
  47. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_integration.py +4 -14
  48. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_meta_output.py +9 -10
  49. euler_eval-2.26.0/tests/test_normal_consistency.py +319 -0
  50. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_points_3d_sparse.py +1 -2
  51. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_rho_a.py +3 -5
  52. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_save_results.py +1 -4
  53. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_validation.py +62 -0
  54. euler_eval-2.22.0/PKG-INFO +0 -698
  55. euler_eval-2.22.0/README.md +0 -660
  56. euler_eval-2.22.0/euler_eval/metrics/normal_consistency.py +0 -222
  57. euler_eval-2.22.0/euler_eval/utils/hierarchy_parser.py +0 -118
  58. euler_eval-2.22.0/euler_eval.egg-info/PKG-INFO +0 -698
  59. euler_eval-2.22.0/pyproject.toml +0 -77
  60. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/calibration.py +0 -0
  61. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/config_paths.py +0 -0
  62. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/depth_standard.py +1 -1
  63. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/lpips_metric.py +0 -0
  64. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/points3d_decomposition.py +0 -0
  65. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/points3d_distance.py +0 -0
  66. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/rgb_lpips.py +0 -0
  67. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval/metrics/rho_a.py +1 -1
  68. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval.egg-info/dependency_links.txt +0 -0
  69. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval.egg-info/entry_points.txt +0 -0
  70. {euler_eval-2.22.0 → euler_eval-2.26.0}/euler_eval.egg-info/top_level.txt +0 -0
  71. {euler_eval-2.22.0 → euler_eval-2.26.0}/init_cache.py +0 -0
  72. {euler_eval-2.22.0 → euler_eval-2.26.0}/setup.cfg +0 -0
  73. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_cli_device.py +0 -0
  74. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_depth_standard.py +0 -0
  75. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_init_cache.py +0 -0
  76. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_metrics_utils.py +0 -0
  77. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_points_3d.py +0 -0
  78. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_rgb_fid_output.py +0 -0
  79. {euler_eval-2.22.0 → euler_eval-2.26.0}/tests/test_sparse_depth.py +0 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Daniel Rothenpieler
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,237 @@
1
+ Metadata-Version: 2.4
2
+ Name: euler-eval
3
+ Version: 2.26.0
4
+ Summary: Evaluation toolkit for depth, RGB, ray and 3D point-map predictions
5
+ Author-email: Daniel Rothenpieler <rothenpielerdaniel@gmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/d-rothen/euler-eval
8
+ Project-URL: Repository, https://github.com/d-rothen/euler-eval
9
+ Project-URL: Issues, https://github.com/d-rothen/euler-eval/issues
10
+ Project-URL: Documentation, https://github.com/d-rothen/euler-eval/tree/main/docs
11
+ Keywords: depth,depth-estimation,evaluation,metrics,benchmark,computer-vision,point-cloud,lidar,3d
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.9
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Programming Language :: Python :: 3.13
22
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
23
+ Classifier: Topic :: Scientific/Engineering :: Image Processing
24
+ Requires-Python: >=3.9
25
+ Description-Content-Type: text/markdown
26
+ License-File: LICENSE
27
+ Requires-Dist: numpy>=1.21.0
28
+ Requires-Dist: scipy>=1.7.0
29
+ Requires-Dist: Pillow>=8.0.0
30
+ Requires-Dist: torch>=1.9.0
31
+ Requires-Dist: torchvision>=0.10.0
32
+ Requires-Dist: opencv-python-headless>=4.5.0
33
+ Requires-Dist: lpips>=0.1.4
34
+ Requires-Dist: torchmetrics>=1.0
35
+ Requires-Dist: tqdm>=4.62.0
36
+ Requires-Dist: euler-loading>=2.15
37
+ Requires-Dist: ds-crawler>=2.10
38
+ Requires-Dist: euler-metric-naming>=0.2
39
+ Provides-Extra: logging
40
+ Requires-Dist: euler-train>=2.11; extra == "logging"
41
+ Provides-Extra: fid
42
+ Requires-Dist: clean-fid>=0.1.35; extra == "fid"
43
+ Provides-Extra: dev
44
+ Requires-Dist: pytest>=7.0.0; extra == "dev"
45
+ Requires-Dist: pytest-cov>=3.0.0; extra == "dev"
46
+ Requires-Dist: black>=22.0.0; extra == "dev"
47
+ Requires-Dist: ruff>=0.2; extra == "dev"
48
+ Dynamic: license-file
49
+
50
+ <!-- euler header — shared across the euler packages.
51
+ Per package, change only: the <h1>, the tagline, and the badge URLs. -->
52
+ <p align="center">
53
+ <img src="https://files.chronodle.com/icons/euler.svg" alt="euler" width="96" height="96">
54
+ </p>
55
+
56
+ <h1 align="center">euler-eval</h1>
57
+
58
+ <p align="center">
59
+ <em>Score depth, RGB, ray and 3D point-map predictions against the same ground truth — from one JSON config.</em>
60
+ </p>
61
+
62
+ <p align="center">
63
+ <a href="https://pypi.org/project/euler-eval/"><img alt="PyPI" src="https://img.shields.io/pypi/v/euler-eval.svg"></a>
64
+ <a href="https://pypi.org/project/euler-eval/"><img alt="Python versions" src="https://img.shields.io/pypi/pyversions/euler-eval.svg"></a>
65
+ <a href="LICENSE"><img alt="License: MIT" src="https://img.shields.io/badge/license-MIT-blue.svg"></a>
66
+ <a href="https://github.com/d-rothen/euler-eval/actions/workflows/ci.yml"><img alt="CI" src="https://github.com/d-rothen/euler-eval/actions/workflows/ci.yml/badge.svg"></a>
67
+ </p>
68
+
69
+ ---
70
+
71
+ Comparing predictions to ground truth is rarely one line of code. The files have
72
+ to be paired, the ground truth may be radial where the prediction is planar, the
73
+ model may only be correct up to scale, and every modality wants a different
74
+ metric set.
75
+
76
+ euler-eval takes a JSON config naming the ground truth and one or more
77
+ prediction datasets, and writes an `eval.json` per modality: files paired by ID,
78
+ decoded from dataset metadata rather than convention, aligned when the
79
+ prediction is not metric, and scored with the metric set that modality deserves.
80
+
81
+ ```mermaid
82
+ flowchart LR
83
+ C["ds-crawler<br/><i>indexes files</i>"] --> H["dataset-head.json<br/><i>euler-dataset-contract</i>"]
84
+ H --> L["euler-loading<br/><i>pairs GT + prediction by ID</i>"]
85
+ L --> E["euler-eval<br/><b>scores the pair</b>"]
86
+ E --> J["eval.json<br/><i>per modality, per dataset</i>"]
87
+ E -.->|optional| T["euler-train<br/><i>experiment logging</i>"]
88
+ ```
89
+
90
+ It is the consuming end of that pipeline and does none of the earlier steps
91
+ itself: a path in the config is a
92
+ [euler-loading](https://github.com/d-rothen/euler-loading) path, and how to
93
+ decode it — loader, units, radial or planar depth, point-cloud column layout —
94
+ is read from the dataset's own index. Metric keys are built with
95
+ [euler-metric-naming](https://github.com/d-rothen/euler-metric-naming), so every
96
+ result is self-describing to downstream tools.
97
+
98
+ ## Install
99
+
100
+ ```bash
101
+ pip install euler-eval # core
102
+ pip install "euler-eval[fid]" # + clean-fid RGB FID backend
103
+ pip install "euler-eval[logging]" # + euler-train experiment logging
104
+ ```
105
+
106
+ Python 3.9 or newer. CUDA is used automatically when available, and every metric
107
+ also runs on CPU.
108
+
109
+ ## Quick start
110
+
111
+ Describe the ground truth and what to score against it:
112
+
113
+ ```json
114
+ {
115
+ "gt": {
116
+ "rgb": { "path": "/data/gt/rgb" },
117
+ "depth": { "path": "/data/gt/depth" }
118
+ },
119
+ "datasets": [
120
+ {
121
+ "name": "model_a",
122
+ "rgb": { "path": "/data/model_a/rgb" },
123
+ "depth": { "path": "/data/model_a/depth" }
124
+ }
125
+ ]
126
+ }
127
+ ```
128
+
129
+ ```bash
130
+ euler-eval config.json --batch-size 32
131
+ ```
132
+
133
+ Ground truth and prediction files are matched by ID rather than by directory
134
+ order, so an incomplete prediction set scores the frames it does have instead of
135
+ silently comparing the wrong pairs. Results are written next to each prediction
136
+ modality:
137
+
138
+ ```json
139
+ {
140
+ "depth": {
141
+ "eval": {
142
+ "metric": {
143
+ "standard": {
144
+ "image_mean": { "absrel": 0.081, "rmse": 3.42, "delta1": 0.93 }
145
+ }
146
+ }
147
+ }
148
+ }
149
+ }
150
+ ```
151
+
152
+ Every path accepts euler-loading's inline selectors, so splits and archives need
153
+ no extra plumbing — `"/data/muses.zip:test#scope=rgb"` reads the `test` split
154
+ straight out of the archive. Full schema: [Configuration](docs/configuration.md).
155
+
156
+ ## What it evaluates
157
+
158
+ | Modality | Config key | Scored against | Headline metrics |
159
+ |---|---|---|---|
160
+ | **Depth** | `depth`, `relative_depth`, `affine_depth` | dense GT depth | `absrel`…`delta3`, PSNR/SSIM/LPIPS/FID/KID, surface-normal consistency, depth-edge F1 |
161
+ | **Sparse depth** | `depth` + `gt.sparse_depth` | a LiDAR-style point cloud, projected into the prediction plane | pointwise depth metrics at projected points, plus directed 3D completeness |
162
+ | **RGB** | `rgb` | GT RGB | PSNR, SSIM, LPIPS, FID, SCE, edge F1, tail errors, HF energy ratio, depth-binned photometric error; optional NIQE/FADE dehazing metrics |
163
+ | **Rays** | `rays` | GT ray direction map | ρ_A (angular-accuracy AUC), angular error, threshold percentages |
164
+ | **Points-3D** | `points_3d` | GT point map, or GT depth unprojected on the fly | 3D EPE/RMSE/δ, radial-vs-lateral decomposition, true-3D normals and edge F1, Chamfer / F-score |
165
+
166
+ A run evaluates whichever modalities are configured on both sides; a depth-only
167
+ prediction is a perfectly ordinary run. The full inventory, with the key each
168
+ metric is written under, is in [Metrics](docs/metrics.md).
169
+
170
+ ## What you get
171
+
172
+ | | |
173
+ |---|---|
174
+ | **ID-based pairing** | GT and prediction frames are matched by file ID and hierarchy, not by filename luck. Calibration and pose files are matched by tree position and shared. |
175
+ | **Metadata-driven decoding** | Radial vs planar depth, value ranges, point-cloud column layout and loader choice all come from the dataset index — no per-dataset flags. |
176
+ | **Honest spaces** | Relative models are scored `native` *and* `metric` (after scale/shift or a similarity gauge), so a scale failure never reads as a geometry failure. |
177
+ | **Benchmark depth bins** | `--benchmark-depth-range MIN MAX` adds near/mid/far bins in square-root depth space, additive to the regular metrics. |
178
+ | **Sky masking** | `--mask-sky` drops sky pixels using GT segmentation — from the metrics *and* from the alignment fit. |
179
+ | **Per-file + aggregate** | Dataset-level numbers plus per-image metrics in dataset-hierarchy order, in the same file. |
180
+ | **Sanity checks** | Results are validated against configurable thresholds; implausible ranges, degenerate inputs and scale mismatches are reported instead of quietly scored. |
181
+ | **Structured metric names** | Each result carries a `metricSet` envelope declaring its namespace, axes, units and metric directions. |
182
+ | **Composable domains** | `--domain dehazing` supplements the stable core set with prediction-only NIQE and FADE; the registry is ready for additional domains. |
183
+ | **euler-train logging** | Optional: register each evaluation in an experiment run with full package provenance. |
184
+
185
+ ## In-training validation
186
+
187
+ The same depth-metric semantics are available for in-memory predictions, so a
188
+ training loop's validation numbers match what the CLI reports later — against
189
+ dense *or* sparse ground truth, without writing predictions to disk:
190
+
191
+ ```python
192
+ from euler_eval import DepthValidationAggregator, evaluate_sparse_depth_sample
193
+
194
+ aggregator = DepthValidationAggregator()
195
+ for sample in val_dataset:
196
+ result = evaluate_sparse_depth_sample(
197
+ model(sample["rgb"]), # (H, W) metres
198
+ sample["sparse_depth"], # (N, C>=3) lidar points
199
+ intrinsics,
200
+ lidar_to_camera,
201
+ alignment="none", # "affine" for relative depth
202
+ )
203
+ aggregator.update(result)
204
+
205
+ aggregator.summary()["standard"]["image_mean"]["absrel"]
206
+ ```
207
+
208
+ Multi-process runs reduce sufficient statistics rather than averages — see
209
+ [In-training validation](docs/validation.md).
210
+
211
+ ## Documentation
212
+
213
+ | Guide | Covers |
214
+ |---|---|
215
+ | [Configuration](docs/configuration.md) | The config file, `gt` and `datasets`, paths and selectors, sparse-depth and points-3d ground truth, euler-train logging |
216
+ | [CLI reference](docs/cli.md) | Every flag, worked examples, device selection, offline cache warmup |
217
+ | [Spaces & alignment](docs/alignment.md) | `native` vs `metric`, depth affine fitting, points-3d gauge alignment, benchmark depth bins |
218
+ | [Metrics](docs/metrics.md) | The full metric inventory per modality |
219
+ | [Results & output](docs/output.md) | Anatomy of `eval.json`, per-file metrics, the sanity-check report |
220
+ | [In-training validation](docs/validation.md) | The programmatic API for scoring in-memory predictions |
221
+
222
+ Start with [the concepts page](docs/README.md) for how the pieces fit together.
223
+
224
+ ## Development
225
+
226
+ ```bash
227
+ git clone https://github.com/d-rothen/euler-eval.git
228
+ cd euler-eval
229
+ uv sync --extra dev # or: pip install -e ".[dev]"
230
+ uv run pytest
231
+ ```
232
+
233
+ See [CONTRIBUTING.md](CONTRIBUTING.md) for the test layout and release process.
234
+
235
+ ## License
236
+
237
+ [MIT](LICENSE) © Daniel Rothenpieler
@@ -0,0 +1,188 @@
1
+ <!-- euler header — shared across the euler packages.
2
+ Per package, change only: the <h1>, the tagline, and the badge URLs. -->
3
+ <p align="center">
4
+ <img src="https://files.chronodle.com/icons/euler.svg" alt="euler" width="96" height="96">
5
+ </p>
6
+
7
+ <h1 align="center">euler-eval</h1>
8
+
9
+ <p align="center">
10
+ <em>Score depth, RGB, ray and 3D point-map predictions against the same ground truth — from one JSON config.</em>
11
+ </p>
12
+
13
+ <p align="center">
14
+ <a href="https://pypi.org/project/euler-eval/"><img alt="PyPI" src="https://img.shields.io/pypi/v/euler-eval.svg"></a>
15
+ <a href="https://pypi.org/project/euler-eval/"><img alt="Python versions" src="https://img.shields.io/pypi/pyversions/euler-eval.svg"></a>
16
+ <a href="LICENSE"><img alt="License: MIT" src="https://img.shields.io/badge/license-MIT-blue.svg"></a>
17
+ <a href="https://github.com/d-rothen/euler-eval/actions/workflows/ci.yml"><img alt="CI" src="https://github.com/d-rothen/euler-eval/actions/workflows/ci.yml/badge.svg"></a>
18
+ </p>
19
+
20
+ ---
21
+
22
+ Comparing predictions to ground truth is rarely one line of code. The files have
23
+ to be paired, the ground truth may be radial where the prediction is planar, the
24
+ model may only be correct up to scale, and every modality wants a different
25
+ metric set.
26
+
27
+ euler-eval takes a JSON config naming the ground truth and one or more
28
+ prediction datasets, and writes an `eval.json` per modality: files paired by ID,
29
+ decoded from dataset metadata rather than convention, aligned when the
30
+ prediction is not metric, and scored with the metric set that modality deserves.
31
+
32
+ ```mermaid
33
+ flowchart LR
34
+ C["ds-crawler<br/><i>indexes files</i>"] --> H["dataset-head.json<br/><i>euler-dataset-contract</i>"]
35
+ H --> L["euler-loading<br/><i>pairs GT + prediction by ID</i>"]
36
+ L --> E["euler-eval<br/><b>scores the pair</b>"]
37
+ E --> J["eval.json<br/><i>per modality, per dataset</i>"]
38
+ E -.->|optional| T["euler-train<br/><i>experiment logging</i>"]
39
+ ```
40
+
41
+ It is the consuming end of that pipeline and does none of the earlier steps
42
+ itself: a path in the config is a
43
+ [euler-loading](https://github.com/d-rothen/euler-loading) path, and how to
44
+ decode it — loader, units, radial or planar depth, point-cloud column layout —
45
+ is read from the dataset's own index. Metric keys are built with
46
+ [euler-metric-naming](https://github.com/d-rothen/euler-metric-naming), so every
47
+ result is self-describing to downstream tools.
48
+
49
+ ## Install
50
+
51
+ ```bash
52
+ pip install euler-eval # core
53
+ pip install "euler-eval[fid]" # + clean-fid RGB FID backend
54
+ pip install "euler-eval[logging]" # + euler-train experiment logging
55
+ ```
56
+
57
+ Python 3.9 or newer. CUDA is used automatically when available, and every metric
58
+ also runs on CPU.
59
+
60
+ ## Quick start
61
+
62
+ Describe the ground truth and what to score against it:
63
+
64
+ ```json
65
+ {
66
+ "gt": {
67
+ "rgb": { "path": "/data/gt/rgb" },
68
+ "depth": { "path": "/data/gt/depth" }
69
+ },
70
+ "datasets": [
71
+ {
72
+ "name": "model_a",
73
+ "rgb": { "path": "/data/model_a/rgb" },
74
+ "depth": { "path": "/data/model_a/depth" }
75
+ }
76
+ ]
77
+ }
78
+ ```
79
+
80
+ ```bash
81
+ euler-eval config.json --batch-size 32
82
+ ```
83
+
84
+ Ground truth and prediction files are matched by ID rather than by directory
85
+ order, so an incomplete prediction set scores the frames it does have instead of
86
+ silently comparing the wrong pairs. Results are written next to each prediction
87
+ modality:
88
+
89
+ ```json
90
+ {
91
+ "depth": {
92
+ "eval": {
93
+ "metric": {
94
+ "standard": {
95
+ "image_mean": { "absrel": 0.081, "rmse": 3.42, "delta1": 0.93 }
96
+ }
97
+ }
98
+ }
99
+ }
100
+ }
101
+ ```
102
+
103
+ Every path accepts euler-loading's inline selectors, so splits and archives need
104
+ no extra plumbing — `"/data/muses.zip:test#scope=rgb"` reads the `test` split
105
+ straight out of the archive. Full schema: [Configuration](docs/configuration.md).
106
+
107
+ ## What it evaluates
108
+
109
+ | Modality | Config key | Scored against | Headline metrics |
110
+ |---|---|---|---|
111
+ | **Depth** | `depth`, `relative_depth`, `affine_depth` | dense GT depth | `absrel`…`delta3`, PSNR/SSIM/LPIPS/FID/KID, surface-normal consistency, depth-edge F1 |
112
+ | **Sparse depth** | `depth` + `gt.sparse_depth` | a LiDAR-style point cloud, projected into the prediction plane | pointwise depth metrics at projected points, plus directed 3D completeness |
113
+ | **RGB** | `rgb` | GT RGB | PSNR, SSIM, LPIPS, FID, SCE, edge F1, tail errors, HF energy ratio, depth-binned photometric error; optional NIQE/FADE dehazing metrics |
114
+ | **Rays** | `rays` | GT ray direction map | ρ_A (angular-accuracy AUC), angular error, threshold percentages |
115
+ | **Points-3D** | `points_3d` | GT point map, or GT depth unprojected on the fly | 3D EPE/RMSE/δ, radial-vs-lateral decomposition, true-3D normals and edge F1, Chamfer / F-score |
116
+
117
+ A run evaluates whichever modalities are configured on both sides; a depth-only
118
+ prediction is a perfectly ordinary run. The full inventory, with the key each
119
+ metric is written under, is in [Metrics](docs/metrics.md).
120
+
121
+ ## What you get
122
+
123
+ | | |
124
+ |---|---|
125
+ | **ID-based pairing** | GT and prediction frames are matched by file ID and hierarchy, not by filename luck. Calibration and pose files are matched by tree position and shared. |
126
+ | **Metadata-driven decoding** | Radial vs planar depth, value ranges, point-cloud column layout and loader choice all come from the dataset index — no per-dataset flags. |
127
+ | **Honest spaces** | Relative models are scored `native` *and* `metric` (after scale/shift or a similarity gauge), so a scale failure never reads as a geometry failure. |
128
+ | **Benchmark depth bins** | `--benchmark-depth-range MIN MAX` adds near/mid/far bins in square-root depth space, additive to the regular metrics. |
129
+ | **Sky masking** | `--mask-sky` drops sky pixels using GT segmentation — from the metrics *and* from the alignment fit. |
130
+ | **Per-file + aggregate** | Dataset-level numbers plus per-image metrics in dataset-hierarchy order, in the same file. |
131
+ | **Sanity checks** | Results are validated against configurable thresholds; implausible ranges, degenerate inputs and scale mismatches are reported instead of quietly scored. |
132
+ | **Structured metric names** | Each result carries a `metricSet` envelope declaring its namespace, axes, units and metric directions. |
133
+ | **Composable domains** | `--domain dehazing` supplements the stable core set with prediction-only NIQE and FADE; the registry is ready for additional domains. |
134
+ | **euler-train logging** | Optional: register each evaluation in an experiment run with full package provenance. |
135
+
136
+ ## In-training validation
137
+
138
+ The same depth-metric semantics are available for in-memory predictions, so a
139
+ training loop's validation numbers match what the CLI reports later — against
140
+ dense *or* sparse ground truth, without writing predictions to disk:
141
+
142
+ ```python
143
+ from euler_eval import DepthValidationAggregator, evaluate_sparse_depth_sample
144
+
145
+ aggregator = DepthValidationAggregator()
146
+ for sample in val_dataset:
147
+ result = evaluate_sparse_depth_sample(
148
+ model(sample["rgb"]), # (H, W) metres
149
+ sample["sparse_depth"], # (N, C>=3) lidar points
150
+ intrinsics,
151
+ lidar_to_camera,
152
+ alignment="none", # "affine" for relative depth
153
+ )
154
+ aggregator.update(result)
155
+
156
+ aggregator.summary()["standard"]["image_mean"]["absrel"]
157
+ ```
158
+
159
+ Multi-process runs reduce sufficient statistics rather than averages — see
160
+ [In-training validation](docs/validation.md).
161
+
162
+ ## Documentation
163
+
164
+ | Guide | Covers |
165
+ |---|---|
166
+ | [Configuration](docs/configuration.md) | The config file, `gt` and `datasets`, paths and selectors, sparse-depth and points-3d ground truth, euler-train logging |
167
+ | [CLI reference](docs/cli.md) | Every flag, worked examples, device selection, offline cache warmup |
168
+ | [Spaces & alignment](docs/alignment.md) | `native` vs `metric`, depth affine fitting, points-3d gauge alignment, benchmark depth bins |
169
+ | [Metrics](docs/metrics.md) | The full metric inventory per modality |
170
+ | [Results & output](docs/output.md) | Anatomy of `eval.json`, per-file metrics, the sanity-check report |
171
+ | [In-training validation](docs/validation.md) | The programmatic API for scoring in-memory predictions |
172
+
173
+ Start with [the concepts page](docs/README.md) for how the pieces fit together.
174
+
175
+ ## Development
176
+
177
+ ```bash
178
+ git clone https://github.com/d-rothen/euler-eval.git
179
+ cd euler-eval
180
+ uv sync --extra dev # or: pip install -e ".[dev]"
181
+ uv run pytest
182
+ ```
183
+
184
+ See [CONTRIBUTING.md](CONTRIBUTING.md) for the test layout and release process.
185
+
186
+ ## License
187
+
188
+ [MIT](LICENSE) © Daniel Rothenpieler
@@ -0,0 +1,83 @@
1
+ # Third-party notices
2
+
3
+ `euler-eval` bundles the pristine natural-scene model parameters from the NIQE
4
+ software release and the foggy/fog-free model parameters from the FADE software
5
+ release. These assets and the corresponding metric implementations are based on
6
+ software published by the Laboratory for Image and Video Engineering (LIVE).
7
+
8
+ ## NIQE software release
9
+
10
+ -----------COPYRIGHT NOTICE STARTS WITH THIS LINE------------
11
+
12
+ Copyright (c) 2011 The University of Texas at Austin
13
+
14
+ All rights reserved.
15
+
16
+ Permission is hereby granted, without written agreement and without license or
17
+ royalty fees, to use, copy, modify, and distribute this code (the source files)
18
+ and its documentation for any purpose, provided that the copyright notice in
19
+ its entirety appear in all copies of this code, and the original source of this
20
+ code, Laboratory for Image and Video Engineering (LIVE,
21
+ http://live.ece.utexas.edu) and Center for Perceptual Systems (CPS,
22
+ http://www.cps.utexas.edu) at the University of Texas at Austin (UT Austin,
23
+ http://www.utexas.edu), is acknowledged in any publication that reports
24
+ research using this code. The research is to be cited in the bibliography as:
25
+
26
+ 1) A. Mittal, R. Soundararajan and A. C. Bovik, "NIQE Software Release",
27
+ URL: http://live.ece.utexas.edu/research/quality/niqe.zip, 2012.
28
+
29
+ 2) A. Mittal, R. Soundararajan and A. C. Bovik, "Making a Completely Blind
30
+ Image Quality Analyzer", submitted to IEEE Signal Processing Letters, 2012.
31
+
32
+ IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT AUSTIN BE LIABLE TO ANY PARTY FOR
33
+ DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES ARISING OUT OF
34
+ THE USE OF THIS DATABASE AND ITS DOCUMENTATION, EVEN IF THE UNIVERSITY OF TEXAS
35
+ AT AUSTIN HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
36
+
37
+ THE UNIVERSITY OF TEXAS AT AUSTIN SPECIFICALLY DISCLAIMS ANY WARRANTIES,
38
+ INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
39
+ FITNESS FOR A PARTICULAR PURPOSE. THE DATABASE PROVIDED HEREUNDER IS ON AN "AS
40
+ IS" BASIS, AND THE UNIVERSITY OF TEXAS AT AUSTIN HAS NO OBLIGATION TO PROVIDE
41
+ MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
42
+
43
+ -----------COPYRIGHT NOTICE ENDS WITH THIS LINE------------
44
+
45
+ ## FADE software release
46
+
47
+ -----------COPYRIGHT NOTICE STARTS WITH THIS LINE------------
48
+
49
+ Copyright (c) 2015 The University of Texas at Austin
50
+
51
+ All rights reserved.
52
+
53
+ Permission is hereby granted, without written agreement and without license or
54
+ royalty fees, to use, copy, modify, and distribute this code (the source files)
55
+ and its documentation for any purpose, provided that the copyright notice in
56
+ its entirety appear in all copies of this code, and the original source of this
57
+ code, Laboratory for Image and Video Engineering (LIVE,
58
+ http://live.ece.utexas.edu) at The University of Texas at Austin (UT Austin,
59
+ http://www.utexas.edu), is acknowledged in any publication that reports
60
+ research using this code. The research is to be cited in the bibliography as:
61
+
62
+ 1. L. K. Choi, J. You, and A. C. Bovik, "Referenceless Prediction of
63
+ Perceptual Fog Density and Perceptual Image Defogging," IEEE Transactions on
64
+ Image Processing, to appear (2015).
65
+
66
+ 2. L. K. Choi, J. You, and A. C. Bovik, "Referenceless perceptual fog density
67
+ prediction model," in Proc. SPIE Human Vis. Electron. Imag., Feb. 2014, 90140H.
68
+
69
+ 3. L. K. Choi, J. You, and A. C. Bovik, "FADE Software Release,"
70
+ URL: http://live.ece.utexas.edu/research/fog/FADE_release.zip, 2015
71
+
72
+ IN NO EVENT SHALL THE UNIVERSITY OF TEXAS AT AUSTIN BE LIABLE TO ANY PARTY FOR
73
+ DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES ARISING OUT OF
74
+ THE USE OF THIS DATABASE AND ITS DOCUMENTATION, EVEN IF THE UNIVERSITY OF TEXAS
75
+ AT AUSTIN HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
76
+
77
+ THE UNIVERSITY OF TEXAS AT AUSTIN SPECIFICALLY DISCLAIMS ANY WARRANTIES,
78
+ INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
79
+ FITNESS FOR A PARTICULAR PURPOSE. THE DATABASE PROVIDED HEREUNDER IS ON AN "AS
80
+ IS" BASIS, AND THE UNIVERSITY OF TEXAS AT AUSTIN HAS NO OBLIGATION TO PROVIDE
81
+ MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
82
+
83
+ -----------COPYRIGHT NOTICE ENDS WITH THIS LINE------------
@@ -18,6 +18,8 @@ from __future__ import annotations
18
18
  from typing import TYPE_CHECKING
19
19
 
20
20
  _VALIDATION_EXPORTS = (
21
+ "BENCHMARK_DEPTH_BIN_NAMES",
22
+ "DepthBenchmarkEvaluation",
21
23
  "DepthSampleEvaluation",
22
24
  "DepthValidationAggregator",
23
25
  "VALIDATION_ALIGNMENT_MODES",
@@ -39,7 +41,9 @@ if TYPE_CHECKING: # pragma: no cover - static import surface for type checkers
39
41
  get_sample_pointcloud_to_camera_extrinsics,
40
42
  )
41
43
  from .validation import ( # noqa: F401
44
+ BENCHMARK_DEPTH_BIN_NAMES,
42
45
  VALIDATION_ALIGNMENT_MODES,
46
+ DepthBenchmarkEvaluation,
43
47
  DepthSampleEvaluation,
44
48
  DepthValidationAggregator,
45
49
  build_validation_gt_dataset,