dataeval 1.0.6__tar.gz → 1.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataeval-1.0.6 → dataeval-1.1.0}/.gitignore +7 -0
- dataeval-1.0.6/README.md → dataeval-1.1.0/PKG-INFO +140 -7
- dataeval-1.0.6/PKG-INFO → dataeval-1.1.0/README.md +77 -62
- {dataeval-1.0.6 → dataeval-1.1.0}/pyproject.toml +145 -59
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/__init__.py +19 -9
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/_embeddings.py +30 -18
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/_experimental.py +8 -14
- dataeval-1.1.0/src/dataeval/_helpers.py +905 -0
- dataeval-1.1.0/src/dataeval/_log.py +81 -0
- dataeval-1.1.0/src/dataeval/_metadata/__init__.py +20 -0
- dataeval-1.1.0/src/dataeval/_metadata/_aggregate.py +230 -0
- dataeval-1.1.0/src/dataeval/_metadata/_columns.py +155 -0
- dataeval-1.1.0/src/dataeval/_metadata/_deprecated.py +392 -0
- dataeval-1.1.0/src/dataeval/_metadata/_encoding.py +297 -0
- dataeval-1.1.0/src/dataeval/_metadata/_entry_legacy.py +341 -0
- dataeval-1.1.0/src/dataeval/_metadata/_filters.py +206 -0
- dataeval-1.1.0/src/dataeval/_metadata/_input.py +133 -0
- dataeval-1.1.0/src/dataeval/_metadata/_keyed.py +151 -0
- dataeval-1.1.0/src/dataeval/_metadata/_links.py +451 -0
- dataeval-1.1.0/src/dataeval/_metadata/_loading.py +190 -0
- dataeval-1.1.0/src/dataeval/_metadata/_metadata.py +3943 -0
- dataeval-1.1.0/src/dataeval/_metadata/_serialize.py +575 -0
- dataeval-1.1.0/src/dataeval/_metadata/_store.py +580 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/__init__.py +81 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_accumulator.py +111 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_base.py +78 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_block.py +67 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_classification.py +132 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_data.py +248 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_dataset.py +92 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_detection.py +138 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_factors.py +309 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_frames.py +24 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_gather.py +95 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_instances.py +60 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_layout.py +98 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_ordering.py +60 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_propagation.py +48 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_reporting.py +59 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_reserved.py +184 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_select.py +119 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_source_index.py +145 -0
- dataeval-1.1.0/src/dataeval/_metadata/_structurers/_tracking.py +388 -0
- dataeval-1.1.0/src/dataeval/_ontology.py +852 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/_version.py +2 -2
- dataeval-1.1.0/src/dataeval/bias/_balance.py +512 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/bias/_diversity.py +55 -27
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/bias/_parity.py +57 -26
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/config.py +75 -2
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/__init__.py +17 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_ber.py +2 -2
- dataeval-1.1.0/src/dataeval/core/_bin.py +607 -0
- dataeval-1.1.0/src/dataeval/core/_calculators/_base.py +216 -0
- dataeval-1.1.0/src/dataeval/core/_calculators/_cache.py +476 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_calculators/_dimensionstats.py +31 -19
- dataeval-1.1.0/src/dataeval/core/_calculators/_hashstats.py +109 -0
- dataeval-1.1.0/src/dataeval/core/_calculators/_pixelstats.py +299 -0
- dataeval-1.1.0/src/dataeval/core/_calculators/_visualstats.py +177 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_clusterer.py +15 -7
- dataeval-1.1.0/src/dataeval/core/_completeness.py +284 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_compute_ratios.py +123 -24
- dataeval-1.1.0/src/dataeval/core/_compute_stats.py +1378 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_coverage.py +20 -12
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_divergence.py +20 -2
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_diversity.py +10 -11
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_fast_hdbscan/_mst.py +11 -6
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_feature_distance.py +4 -5
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_hash.py +35 -10
- dataeval-1.1.0/src/dataeval/core/_label_alignment.py +271 -0
- dataeval-1.1.0/src/dataeval/core/_label_coverage.py +246 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_label_errors.py +2 -2
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_label_parity.py +14 -5
- dataeval-1.1.0/src/dataeval/core/_label_reconciliation.py +147 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_label_stats.py +2 -2
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_metadata_insights.py +2 -2
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_mst.py +17 -17
- dataeval-1.1.0/src/dataeval/core/_mutual_info.py +747 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_nullmodel.py +317 -7
- dataeval-1.1.0/src/dataeval/core/_ontology_validation.py +206 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_parity.py +6 -6
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_rank.py +51 -2
- dataeval-1.1.0/src/dataeval/core/_track_stats.py +501 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_uap.py +8 -2
- dataeval-1.1.0/src/dataeval/data/__init__.py +42 -0
- {dataeval-1.0.6/src/dataeval/selection → dataeval-1.1.0/src/dataeval/data}/_classbalance.py +22 -19
- dataeval-1.1.0/src/dataeval/data/_classfilter.py +71 -0
- dataeval-1.1.0/src/dataeval/data/_crop.py +107 -0
- dataeval-1.1.0/src/dataeval/data/_crops.py +404 -0
- dataeval-1.1.0/src/dataeval/data/_geometry.py +151 -0
- {dataeval-1.0.6/src/dataeval/selection → dataeval-1.1.0/src/dataeval/data}/_indices.py +5 -6
- dataeval-1.1.0/src/dataeval/data/_invalidates.py +105 -0
- dataeval-1.1.0/src/dataeval/data/_limit.py +27 -0
- dataeval-1.1.0/src/dataeval/data/_merge.py +163 -0
- dataeval-1.1.0/src/dataeval/data/_relabel.py +209 -0
- dataeval-1.1.0/src/dataeval/data/_resize.py +240 -0
- dataeval-1.1.0/src/dataeval/data/_reverse.py +12 -0
- dataeval-1.1.0/src/dataeval/data/_selectchannels.py +171 -0
- {dataeval-1.0.6/src/dataeval/selection → dataeval-1.1.0/src/dataeval/data}/_shuffle.py +5 -6
- dataeval-1.0.6/src/dataeval/utils/data.py → dataeval-1.1.0/src/dataeval/data/_split.py +110 -98
- dataeval-1.1.0/src/dataeval/data/_torchvision.py +265 -0
- dataeval-1.1.0/src/dataeval/data/_tracks.py +85 -0
- dataeval-1.1.0/src/dataeval/data/_unzip.py +84 -0
- dataeval-1.1.0/src/dataeval/data/_view.py +285 -0
- dataeval-1.1.0/src/dataeval/exceptions.py +118 -0
- dataeval-1.1.0/src/dataeval/extractors/__init__.py +31 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/extractors/_bovw.py +42 -17
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/extractors/_flatten.py +9 -3
- dataeval-1.1.0/src/dataeval/extractors/_geometry.py +78 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/extractors/_onnx.py +60 -21
- dataeval-1.1.0/src/dataeval/extractors/_scores.py +66 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/extractors/_torch.py +81 -22
- dataeval-1.1.0/src/dataeval/extractors/_uncertainty.py +507 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/flags.py +36 -13
- dataeval-1.1.0/src/dataeval/models/__init__.py +20 -0
- dataeval-1.1.0/src/dataeval/models/_backends.py +174 -0
- dataeval-1.1.0/src/dataeval/models/_input.py +129 -0
- dataeval-1.1.0/src/dataeval/models/_metadata.py +140 -0
- dataeval-1.1.0/src/dataeval/models/_predictors.py +461 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/performance/_output.py +3 -2
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/performance/_sufficiency.py +25 -12
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/protocols.py +449 -142
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/quality/_duplicates.py +85 -61
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/quality/_outliers.py +97 -31
- dataeval-1.1.0/src/dataeval/quality/_shared.py +199 -0
- dataeval-1.1.0/src/dataeval/scope/__init__.py +14 -0
- dataeval-1.1.0/src/dataeval/scope/_coverage.py +480 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/scope/_prioritize.py +64 -40
- dataeval-1.1.0/src/dataeval/scope/_representation.py +364 -0
- dataeval-1.1.0/src/dataeval/selection/__init__.py +71 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/__init__.py +2 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_drift/_base.py +29 -8
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_drift/_chunk.py +1 -1
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_drift/_domain_classifier.py +3 -3
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_drift/_kneighbors.py +5 -5
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_drift/_mmd.py +30 -8
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_drift/_reconstruction.py +45 -7
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_drift/_univariate.py +41 -14
- dataeval-1.1.0/src/dataeval/shift/_drift/_wasserstein.py +431 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_ood/_base.py +6 -2
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_ood/_reconstruction.py +36 -5
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_shared/_reconstruction.py +3 -3
- dataeval-1.1.0/src/dataeval/types/__init__.py +67 -0
- dataeval-1.1.0/src/dataeval/types/_array.py +34 -0
- dataeval-1.1.0/src/dataeval/types/_config.py +32 -0
- dataeval-1.1.0/src/dataeval/types/_evaluator.py +108 -0
- dataeval-1.1.0/src/dataeval/types/_execution.py +70 -0
- dataeval-1.1.0/src/dataeval/types/_factors.py +632 -0
- dataeval-1.1.0/src/dataeval/types/_index.py +92 -0
- dataeval-1.1.0/src/dataeval/types/_ontology.py +108 -0
- dataeval-1.1.0/src/dataeval/types/_output.py +292 -0
- dataeval-1.1.0/src/dataeval/types/_schema.py +256 -0
- dataeval-1.1.0/src/dataeval/types/_track.py +40 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/utils/__init__.py +1 -2
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/utils/_internal.py +111 -4
- dataeval-1.1.0/src/dataeval/utils/_validate.py +272 -0
- dataeval-1.1.0/src/dataeval/utils/data.py +34 -0
- dataeval-1.1.0/src/dataeval/utils/preprocessing.py +1244 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/utils/training.py +103 -18
- dataeval-1.0.6/src/dataeval/_helpers.py +0 -47
- dataeval-1.0.6/src/dataeval/_log.py +0 -16
- dataeval-1.0.6/src/dataeval/_metadata.py +0 -1640
- dataeval-1.0.6/src/dataeval/_warm_cache.py +0 -129
- dataeval-1.0.6/src/dataeval/bias/_balance.py +0 -358
- dataeval-1.0.6/src/dataeval/core/_bin.py +0 -196
- dataeval-1.0.6/src/dataeval/core/_calculators/_base.py +0 -113
- dataeval-1.0.6/src/dataeval/core/_calculators/_cache.py +0 -75
- dataeval-1.0.6/src/dataeval/core/_calculators/_hashstats.py +0 -84
- dataeval-1.0.6/src/dataeval/core/_calculators/_pixelstats.py +0 -193
- dataeval-1.0.6/src/dataeval/core/_calculators/_visualstats.py +0 -97
- dataeval-1.0.6/src/dataeval/core/_completeness.py +0 -168
- dataeval-1.0.6/src/dataeval/core/_compute_stats.py +0 -604
- dataeval-1.0.6/src/dataeval/core/_mutual_info.py +0 -322
- dataeval-1.0.6/src/dataeval/exceptions.py +0 -41
- dataeval-1.0.6/src/dataeval/extractors/__init__.py +0 -15
- dataeval-1.0.6/src/dataeval/extractors/_uncertainty.py +0 -245
- dataeval-1.0.6/src/dataeval/quality/_shared.py +0 -112
- dataeval-1.0.6/src/dataeval/scope/__init__.py +0 -10
- dataeval-1.0.6/src/dataeval/selection/__init__.py +0 -20
- dataeval-1.0.6/src/dataeval/selection/_classfilter.py +0 -106
- dataeval-1.0.6/src/dataeval/selection/_limit.py +0 -24
- dataeval-1.0.6/src/dataeval/selection/_reverse.py +0 -14
- dataeval-1.0.6/src/dataeval/selection/_select.py +0 -185
- dataeval-1.0.6/src/dataeval/types.py +0 -540
- dataeval-1.0.6/src/dataeval/utils/preprocessing.py +0 -611
- {dataeval-1.0.6 → dataeval-1.1.0}/LICENSE +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/bias/__init__.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_calculators/__init__.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_calculators/_register.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_calculators/_registry.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_fast_hdbscan/_cluster_trees.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/core/_fast_hdbscan/_disjoint_set.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/performance/__init__.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/performance/_aggregator.py +8 -8
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/performance/schedules.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/py.typed +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/quality/__init__.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_drift/__init__.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_ood/__init__.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_ood/_domain_classifier.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_ood/_kneighbors.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_shared/__init__.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_shared/_domain_classifier.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/_shared/_kneighbors.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/shift/update_strategies.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/utils/losses.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/utils/models.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/utils/onnx.py +0 -0
- {dataeval-1.0.6 → dataeval-1.1.0}/src/dataeval/utils/thresholds.py +0 -0
|
@@ -16,6 +16,12 @@ docs/source/tutorials/notebooks/checkpoints/
|
|
|
16
16
|
# See scripts/README.md for details
|
|
17
17
|
.jupyter_cache/
|
|
18
18
|
|
|
19
|
+
# IPython runtime files and configuration
|
|
20
|
+
docs/source/.ipython/*
|
|
21
|
+
!docs/source/.ipython/profile_default/
|
|
22
|
+
docs/source/.ipython/profile_default/*
|
|
23
|
+
!docs/source/.ipython/profile_default/ipython_kernel_config.py
|
|
24
|
+
|
|
19
25
|
# Generated notebook files - stored in docs-artifacts branches for Colab
|
|
20
26
|
docs/source/notebooks/*.ipynb
|
|
21
27
|
|
|
@@ -24,6 +30,7 @@ output/
|
|
|
24
30
|
|
|
25
31
|
.tox/
|
|
26
32
|
.nox/
|
|
33
|
+
.numba-cache/
|
|
27
34
|
.python-version
|
|
28
35
|
.cuda-version
|
|
29
36
|
|
|
@@ -1,4 +1,68 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: dataeval
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: DataEval provides a simple interface to characterize image data and its impact on model performance across classification and object-detection tasks
|
|
5
|
+
Project-URL: Homepage, https://dataeval.ai/
|
|
6
|
+
Project-URL: Repository, https://github.com/aria-ml/dataeval/
|
|
7
|
+
Project-URL: Documentation, https://dataeval.readthedocs.io/
|
|
8
|
+
Author-email: Andrew Weng <andrew.weng@ariacoustics.com>, Bill Peria <bill.peria@ariacoustics.com>, Christina Doty <christina.doty@ariacoustics.com>, Jon Botts <jonathan.botts@ariacoustics.com>, Jonathan Christian <jonathan.christian@ariacoustics.com>, Justin McMillan <justin.mcmillan@ariacoustics.com>, Ryan Wood <ryan.wood@ariacoustics.com>, Scott Swan <scott.swan@ariacoustics.com>
|
|
9
|
+
Maintainer-email: ARiA <dataeval@ariacoustics.com>
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Requires-Dist: lightgbm>=4
|
|
24
|
+
Requires-Dist: maite>=0.9.4
|
|
25
|
+
Requires-Dist: numba>=0.61.0
|
|
26
|
+
Requires-Dist: numpy>=1.24.2
|
|
27
|
+
Requires-Dist: polars>=1.0.0
|
|
28
|
+
Requires-Dist: psutil>=7.0.0
|
|
29
|
+
Requires-Dist: pydantic>=2.8
|
|
30
|
+
Requires-Dist: scikit-learn>=1.5.0
|
|
31
|
+
Requires-Dist: scipy>=1.10.0
|
|
32
|
+
Requires-Dist: torch>=2.2.0
|
|
33
|
+
Requires-Dist: typing-extensions>=4.12
|
|
34
|
+
Requires-Dist: xxhash>=3.4
|
|
35
|
+
Provides-Extra: cpu
|
|
36
|
+
Requires-Dist: torch>=2.2.0; extra == 'cpu'
|
|
37
|
+
Requires-Dist: torchvision>=0.17.0; extra == 'cpu'
|
|
38
|
+
Provides-Extra: cu126
|
|
39
|
+
Requires-Dist: torch>=2.2.0; extra == 'cu126'
|
|
40
|
+
Requires-Dist: torchvision>=0.17.0; extra == 'cu126'
|
|
41
|
+
Provides-Extra: cu130
|
|
42
|
+
Requires-Dist: torch>=2.2.0; extra == 'cu130'
|
|
43
|
+
Requires-Dist: torchvision>=0.17.0; extra == 'cu130'
|
|
44
|
+
Provides-Extra: litert
|
|
45
|
+
Requires-Dist: ai-edge-litert>=2.0; (python_version <= '3.14') and extra == 'litert'
|
|
46
|
+
Provides-Extra: onnx
|
|
47
|
+
Requires-Dist: onnx>=1.14.0; extra == 'onnx'
|
|
48
|
+
Requires-Dist: onnxruntime>=1.17; extra == 'onnx'
|
|
49
|
+
Provides-Extra: onnx-cu126
|
|
50
|
+
Requires-Dist: onnx>=1.14.0; extra == 'onnx-cu126'
|
|
51
|
+
Requires-Dist: onnxruntime-gpu<1.24,>=1.17; (python_version == '3.10' and extra != 'onnx-cu130') and extra == 'onnx-cu126'
|
|
52
|
+
Requires-Dist: onnxruntime-gpu<1.27,>=1.17; (python_version >= '3.11' and extra != 'onnx-cu130') and extra == 'onnx-cu126'
|
|
53
|
+
Requires-Dist: onnxruntime-gpu<1.27,>=1.24; (python_version >= '3.14' and extra != 'onnx-cu130') and extra == 'onnx-cu126'
|
|
54
|
+
Provides-Extra: onnx-cu130
|
|
55
|
+
Requires-Dist: onnx>=1.14.0; extra == 'onnx-cu130'
|
|
56
|
+
Requires-Dist: onnxruntime-gpu>=1.27; (python_version >= '3.11' and extra != 'onnx-cu126') and extra == 'onnx-cu130'
|
|
57
|
+
Requires-Dist: onnxruntime>=1.17; (python_version == '3.10' and extra != 'onnx-cu126') and extra == 'onnx-cu130'
|
|
58
|
+
Provides-Extra: ontology
|
|
59
|
+
Requires-Dist: rdflib>=7.0; extra == 'ontology'
|
|
60
|
+
Provides-Extra: opencv
|
|
61
|
+
Requires-Dist: opencv-python-headless>=4.8.0; extra == 'opencv'
|
|
62
|
+
Description-Content-Type: text/markdown
|
|
63
|
+
|
|
1
64
|
<!-- markdownlint-disable MD041 -->
|
|
65
|
+
|
|
2
66
|

|
|
3
67
|
|
|
4
68
|
<!-- :auto badges: -->
|
|
@@ -64,6 +128,7 @@ Choose your preferred method of installation below or follow our
|
|
|
64
128
|
[installation guide](docs/source/getting-started/installation.md).
|
|
65
129
|
|
|
66
130
|
- [Installing with pip](#installing-with-pip)
|
|
131
|
+
- [Installing with uv](#installing-with-uv)
|
|
67
132
|
- [Installing with conda/mamba](#installing-with-conda)
|
|
68
133
|
- [Installing from GitHub](#installing-from-github)
|
|
69
134
|
|
|
@@ -75,14 +140,76 @@ You can install DataEval directly from pypi.org using the following command.
|
|
|
75
140
|
pip install dataeval
|
|
76
141
|
```
|
|
77
142
|
|
|
143
|
+
By default, PyTorch is installed from PyPI, which bundles CUDA support on Linux
|
|
144
|
+
and is a much larger download than the CPU build. To choose a specific PyTorch
|
|
145
|
+
variant, install `torch` from that variant's wheel index **first**, then install
|
|
146
|
+
DataEval — it accepts the build already present in the environment:
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
# 1. Pick your PyTorch build (cpu / cu126 / cu130)
|
|
150
|
+
pip install torch --index-url https://download.pytorch.org/whl/cu130
|
|
151
|
+
|
|
152
|
+
# 2. Install DataEval
|
|
153
|
+
pip install dataeval
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
> **Use `--index-url`, not `--extra-index-url`, to pick a CUDA build.**
|
|
157
|
+
> `--extra-index-url` *adds* an index instead of replacing PyPI, and pip then
|
|
158
|
+
> takes the highest version across both. The CUDA indexes lag the latest PyTorch
|
|
159
|
+
> release, so PyPI usually wins and you silently get the default CUDA-bundled
|
|
160
|
+
> build — the install succeeds with no warning. `--index-url` replaces the index
|
|
161
|
+
> outright, so it is reliable. (For CPU only,
|
|
162
|
+
> `pip install dataeval --extra-index-url https://download.pytorch.org/whl/cpu`
|
|
163
|
+
> does work, because the CPU index tracks the latest release.)
|
|
164
|
+
>
|
|
165
|
+
> **The `cpu` / `cu126` / `cu130` extras do not select a PyTorch variant under
|
|
166
|
+
> pip.** All three declare the same requirements (`torch`, `torchvision`); what
|
|
167
|
+
> distinguishes them is `[tool.uv.sources]`, which routes those packages to the
|
|
168
|
+
> right wheel index. That is project metadata applied by uv when resolving **from
|
|
169
|
+
> source** — it is not part of the published wheel. Select the variant with
|
|
170
|
+
> `--index-url` under pip, `--torch-backend` under `uv pip`, and use the extras
|
|
171
|
+
> only for source installs.
|
|
172
|
+
|
|
173
|
+
### **torchvision (optional)**
|
|
174
|
+
|
|
175
|
+
`torchvision` is not a DataEval dependency, and nothing imports it until you reach
|
|
176
|
+
for `TorchvisionTransform` — the escape hatch for running a torchvision v2
|
|
177
|
+
transform across a dataset view. If you want that class, install torchvision
|
|
178
|
+
yourself, from the **same index as your torch build**:
|
|
179
|
+
|
|
180
|
+
```bash
|
|
181
|
+
pip install torchvision --index-url https://download.pytorch.org/whl/cu130
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
> **Do not mix indexes.** A `torchvision` from PyPI alongside a torch installed
|
|
185
|
+
> from a wheel index resolves and installs cleanly, then fails on
|
|
186
|
+
> `import torchvision` with
|
|
187
|
+
> `RuntimeError: operator torchvision::nms does not exist`. torchvision's compiled
|
|
188
|
+
> ops are built against one specific torch build, so both packages must come from
|
|
189
|
+
> the same index — which is also why `pip install dataeval[cpu]` is the wrong way
|
|
190
|
+
> to obtain it.
|
|
191
|
+
|
|
192
|
+
### **Installing with uv**
|
|
193
|
+
|
|
194
|
+
```bash
|
|
195
|
+
uv pip install dataeval --torch-backend cpu # or cu126 / cu130 / auto
|
|
196
|
+
```
|
|
197
|
+
|
|
78
198
|
### **Installing with conda**
|
|
79
199
|
|
|
80
|
-
DataEval can be installed
|
|
81
|
-
`environment.yaml` file. As some dependencies are installed from the `pytorch`
|
|
82
|
-
channel, the channel is specified in the below example.
|
|
200
|
+
DataEval can be installed from conda-forge:
|
|
83
201
|
|
|
84
202
|
```bash
|
|
85
|
-
|
|
203
|
+
conda install -c conda-forge dataeval
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
Alternatively, create an environment from the provided `environment.yml` at the
|
|
207
|
+
repository root. PyTorch is installed into that environment from PyPI via `pip`,
|
|
208
|
+
so this path gives you the CPU/CUDA-bundled PyPI build rather than a specific
|
|
209
|
+
variant.
|
|
210
|
+
|
|
211
|
+
```bash
|
|
212
|
+
micromamba create -f environment.yml
|
|
86
213
|
```
|
|
87
214
|
|
|
88
215
|
### **Installing from GitHub**
|
|
@@ -95,6 +222,11 @@ git clone https://github.com/aria-ml/dataeval.git
|
|
|
95
222
|
cd dataeval
|
|
96
223
|
```
|
|
97
224
|
|
|
225
|
+
> **Contributing rather than just installing?** Use
|
|
226
|
+
> `uvx --with nox-uv nox -s dev` to build a full development environment — tests,
|
|
227
|
+
> linting, type checking and docs tooling included. See
|
|
228
|
+
> [Development Setup](./CONTRIBUTING.md#development-setup) for the options it takes.
|
|
229
|
+
|
|
98
230
|
#### **Using Poetry**
|
|
99
231
|
|
|
100
232
|
Install DataEval.
|
|
@@ -346,7 +478,7 @@ shape: (3, 5)
|
|
|
346
478
|
|
|
347
479
|
A result with many large groups is a signal that your dataset contains
|
|
348
480
|
repeated collection events. Before training, remove all but one sample from
|
|
349
|
-
each group. See the [deduplication how-to guide](./docs/source/notebooks/h2_deduplicate.
|
|
481
|
+
each group. See the [deduplication how-to guide](./docs/source/notebooks/h2_deduplicate.py)
|
|
350
482
|
for a complete walkthrough, including how to choose which sample to keep.
|
|
351
483
|
|
|
352
484
|
### Where to go next
|
|
@@ -354,8 +486,9 @@ for a complete walkthrough, including how to choose which sample to keep.
|
|
|
354
486
|
Not sure what to evaluate first? Use the [Which tool should I use?](./docs/source/getting-started/which-tool.md)
|
|
355
487
|
guide to find the right evaluator for your situation.
|
|
356
488
|
|
|
357
|
-
Know which tool to use, then check out
|
|
358
|
-
for a quick-reference table of
|
|
489
|
+
Know which tool to use, then check out [What data does each tool need?](./docs/source/getting-started/input-requirements.md)
|
|
490
|
+
for a quick-reference table of every algorithm's inputs, and the
|
|
491
|
+
[Functional Overview](./docs/source/reference/FunctionalOverview.md) for task applicability.
|
|
359
492
|
|
|
360
493
|
Want to just explore the documentation? The [Where to go next](./docs/source/getting-started/where-to-go-next.md)
|
|
361
494
|
page allows you to jump around between the different areas of the documentation with small summaries of what each page covers.
|
|
@@ -1,59 +1,5 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: dataeval
|
|
3
|
-
Version: 1.0.6
|
|
4
|
-
Summary: DataEval provides a simple interface to characterize image data and its impact on model performance across classification and object-detection tasks
|
|
5
|
-
Project-URL: Homepage, https://dataeval.ai/
|
|
6
|
-
Project-URL: Repository, https://github.com/aria-ml/dataeval/
|
|
7
|
-
Project-URL: Documentation, https://dataeval.readthedocs.io/
|
|
8
|
-
Author-email: Andrew Weng <andrew.weng@ariacoustics.com>, Bill Peria <bill.peria@ariacoustics.com>, Christina Doty <christina.doty@ariacoustics.com>, Jon Botts <jonathan.botts@ariacoustics.com>, Jonathan Christian <jonathan.christian@ariacoustics.com>, Justin McMillan <justin.mcmillan@ariacoustics.com>, Ryan Wood <ryan.wood@ariacoustics.com>, Scott Swan <scott.swan@ariacoustics.com>
|
|
9
|
-
Maintainer-email: ARiA <dataeval@ariacoustics.com>
|
|
10
|
-
License-Expression: MIT
|
|
11
|
-
License-File: LICENSE
|
|
12
|
-
Classifier: Development Status :: 5 - Production/Stable
|
|
13
|
-
Classifier: Intended Audience :: Science/Research
|
|
14
|
-
Classifier: Operating System :: OS Independent
|
|
15
|
-
Classifier: Programming Language :: Python :: 3 :: Only
|
|
16
|
-
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
-
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
-
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
-
Classifier: Programming Language :: Python :: 3.14
|
|
21
|
-
Classifier: Topic :: Scientific/Engineering
|
|
22
|
-
Requires-Python: >=3.10
|
|
23
|
-
Requires-Dist: lightgbm>=4
|
|
24
|
-
Requires-Dist: numba>=0.61.0
|
|
25
|
-
Requires-Dist: numpy>=1.24.2
|
|
26
|
-
Requires-Dist: polars>=1.0.0
|
|
27
|
-
Requires-Dist: psutil>=7.0.0
|
|
28
|
-
Requires-Dist: pydantic>=2.8
|
|
29
|
-
Requires-Dist: scikit-learn>=1.5.0
|
|
30
|
-
Requires-Dist: scipy>=1.10.0
|
|
31
|
-
Requires-Dist: torch>=2.2.0
|
|
32
|
-
Requires-Dist: typing-extensions>=4.12
|
|
33
|
-
Requires-Dist: xxhash>=3.4
|
|
34
|
-
Provides-Extra: cpu
|
|
35
|
-
Requires-Dist: torch>=2.2.0; extra == 'cpu'
|
|
36
|
-
Requires-Dist: torchvision>=0.17.0; extra == 'cpu'
|
|
37
|
-
Provides-Extra: cu118
|
|
38
|
-
Requires-Dist: torch>=2.2.0; extra == 'cu118'
|
|
39
|
-
Requires-Dist: torchvision>=0.17.0; extra == 'cu118'
|
|
40
|
-
Provides-Extra: cu124
|
|
41
|
-
Requires-Dist: torch>=2.2.0; extra == 'cu124'
|
|
42
|
-
Requires-Dist: torchvision>=0.17.0; extra == 'cu124'
|
|
43
|
-
Provides-Extra: cu128
|
|
44
|
-
Requires-Dist: torch>=2.2.0; extra == 'cu128'
|
|
45
|
-
Requires-Dist: torchvision>=0.17.0; extra == 'cu128'
|
|
46
|
-
Provides-Extra: onnx
|
|
47
|
-
Requires-Dist: onnx; extra == 'onnx'
|
|
48
|
-
Requires-Dist: onnxruntime>=1.14.0; extra == 'onnx'
|
|
49
|
-
Provides-Extra: onnx-gpu
|
|
50
|
-
Requires-Dist: onnx; extra == 'onnx-gpu'
|
|
51
|
-
Requires-Dist: onnxruntime-gpu>=1.14.0; extra == 'onnx-gpu'
|
|
52
|
-
Provides-Extra: opencv
|
|
53
|
-
Requires-Dist: opencv-python-headless>=4.8.0; extra == 'opencv'
|
|
54
|
-
Description-Content-Type: text/markdown
|
|
55
|
-
|
|
56
1
|
<!-- markdownlint-disable MD041 -->
|
|
2
|
+
|
|
57
3
|

|
|
58
4
|
|
|
59
5
|
<!-- :auto badges: -->
|
|
@@ -119,6 +65,7 @@ Choose your preferred method of installation below or follow our
|
|
|
119
65
|
[installation guide](docs/source/getting-started/installation.md).
|
|
120
66
|
|
|
121
67
|
- [Installing with pip](#installing-with-pip)
|
|
68
|
+
- [Installing with uv](#installing-with-uv)
|
|
122
69
|
- [Installing with conda/mamba](#installing-with-conda)
|
|
123
70
|
- [Installing from GitHub](#installing-from-github)
|
|
124
71
|
|
|
@@ -130,14 +77,76 @@ You can install DataEval directly from pypi.org using the following command.
|
|
|
130
77
|
pip install dataeval
|
|
131
78
|
```
|
|
132
79
|
|
|
80
|
+
By default, PyTorch is installed from PyPI, which bundles CUDA support on Linux
|
|
81
|
+
and is a much larger download than the CPU build. To choose a specific PyTorch
|
|
82
|
+
variant, install `torch` from that variant's wheel index **first**, then install
|
|
83
|
+
DataEval — it accepts the build already present in the environment:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
# 1. Pick your PyTorch build (cpu / cu126 / cu130)
|
|
87
|
+
pip install torch --index-url https://download.pytorch.org/whl/cu130
|
|
88
|
+
|
|
89
|
+
# 2. Install DataEval
|
|
90
|
+
pip install dataeval
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
> **Use `--index-url`, not `--extra-index-url`, to pick a CUDA build.**
|
|
94
|
+
> `--extra-index-url` *adds* an index instead of replacing PyPI, and pip then
|
|
95
|
+
> takes the highest version across both. The CUDA indexes lag the latest PyTorch
|
|
96
|
+
> release, so PyPI usually wins and you silently get the default CUDA-bundled
|
|
97
|
+
> build — the install succeeds with no warning. `--index-url` replaces the index
|
|
98
|
+
> outright, so it is reliable. (For CPU only,
|
|
99
|
+
> `pip install dataeval --extra-index-url https://download.pytorch.org/whl/cpu`
|
|
100
|
+
> does work, because the CPU index tracks the latest release.)
|
|
101
|
+
>
|
|
102
|
+
> **The `cpu` / `cu126` / `cu130` extras do not select a PyTorch variant under
|
|
103
|
+
> pip.** All three declare the same requirements (`torch`, `torchvision`); what
|
|
104
|
+
> distinguishes them is `[tool.uv.sources]`, which routes those packages to the
|
|
105
|
+
> right wheel index. That is project metadata applied by uv when resolving **from
|
|
106
|
+
> source** — it is not part of the published wheel. Select the variant with
|
|
107
|
+
> `--index-url` under pip, `--torch-backend` under `uv pip`, and use the extras
|
|
108
|
+
> only for source installs.
|
|
109
|
+
|
|
110
|
+
### **torchvision (optional)**
|
|
111
|
+
|
|
112
|
+
`torchvision` is not a DataEval dependency, and nothing imports it until you reach
|
|
113
|
+
for `TorchvisionTransform` — the escape hatch for running a torchvision v2
|
|
114
|
+
transform across a dataset view. If you want that class, install torchvision
|
|
115
|
+
yourself, from the **same index as your torch build**:
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
pip install torchvision --index-url https://download.pytorch.org/whl/cu130
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
> **Do not mix indexes.** A `torchvision` from PyPI alongside a torch installed
|
|
122
|
+
> from a wheel index resolves and installs cleanly, then fails on
|
|
123
|
+
> `import torchvision` with
|
|
124
|
+
> `RuntimeError: operator torchvision::nms does not exist`. torchvision's compiled
|
|
125
|
+
> ops are built against one specific torch build, so both packages must come from
|
|
126
|
+
> the same index — which is also why `pip install dataeval[cpu]` is the wrong way
|
|
127
|
+
> to obtain it.
|
|
128
|
+
|
|
129
|
+
### **Installing with uv**
|
|
130
|
+
|
|
131
|
+
```bash
|
|
132
|
+
uv pip install dataeval --torch-backend cpu # or cu126 / cu130 / auto
|
|
133
|
+
```
|
|
134
|
+
|
|
133
135
|
### **Installing with conda**
|
|
134
136
|
|
|
135
|
-
DataEval can be installed
|
|
136
|
-
`environment.yaml` file. As some dependencies are installed from the `pytorch`
|
|
137
|
-
channel, the channel is specified in the below example.
|
|
137
|
+
DataEval can be installed from conda-forge:
|
|
138
138
|
|
|
139
139
|
```bash
|
|
140
|
-
|
|
140
|
+
conda install -c conda-forge dataeval
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
Alternatively, create an environment from the provided `environment.yml` at the
|
|
144
|
+
repository root. PyTorch is installed into that environment from PyPI via `pip`,
|
|
145
|
+
so this path gives you the CPU/CUDA-bundled PyPI build rather than a specific
|
|
146
|
+
variant.
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
micromamba create -f environment.yml
|
|
141
150
|
```
|
|
142
151
|
|
|
143
152
|
### **Installing from GitHub**
|
|
@@ -150,6 +159,11 @@ git clone https://github.com/aria-ml/dataeval.git
|
|
|
150
159
|
cd dataeval
|
|
151
160
|
```
|
|
152
161
|
|
|
162
|
+
> **Contributing rather than just installing?** Use
|
|
163
|
+
> `uvx --with nox-uv nox -s dev` to build a full development environment — tests,
|
|
164
|
+
> linting, type checking and docs tooling included. See
|
|
165
|
+
> [Development Setup](./CONTRIBUTING.md#development-setup) for the options it takes.
|
|
166
|
+
|
|
153
167
|
#### **Using Poetry**
|
|
154
168
|
|
|
155
169
|
Install DataEval.
|
|
@@ -401,7 +415,7 @@ shape: (3, 5)
|
|
|
401
415
|
|
|
402
416
|
A result with many large groups is a signal that your dataset contains
|
|
403
417
|
repeated collection events. Before training, remove all but one sample from
|
|
404
|
-
each group. See the [deduplication how-to guide](./docs/source/notebooks/h2_deduplicate.
|
|
418
|
+
each group. See the [deduplication how-to guide](./docs/source/notebooks/h2_deduplicate.py)
|
|
405
419
|
for a complete walkthrough, including how to choose which sample to keep.
|
|
406
420
|
|
|
407
421
|
### Where to go next
|
|
@@ -409,8 +423,9 @@ for a complete walkthrough, including how to choose which sample to keep.
|
|
|
409
423
|
Not sure what to evaluate first? Use the [Which tool should I use?](./docs/source/getting-started/which-tool.md)
|
|
410
424
|
guide to find the right evaluator for your situation.
|
|
411
425
|
|
|
412
|
-
Know which tool to use, then check out
|
|
413
|
-
for a quick-reference table of
|
|
426
|
+
Know which tool to use, then check out [What data does each tool need?](./docs/source/getting-started/input-requirements.md)
|
|
427
|
+
for a quick-reference table of every algorithm's inputs, and the
|
|
428
|
+
[Functional Overview](./docs/source/reference/FunctionalOverview.md) for task applicability.
|
|
414
429
|
|
|
415
430
|
Want to just explore the documentation? The [Where to go next](./docs/source/getting-started/where-to-go-next.md)
|
|
416
431
|
page allows you to jump around between the different areas of the documentation with small summaries of what each page covers.
|