dataeval 1.1.0rc6__tar.gz → 1.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/PKG-INFO +22 -19
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/README.md +5 -5
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/pyproject.toml +50 -25
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_columns.py +26 -2
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_keyed.py +20 -4
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_metadata.py +98 -8
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_reserved.py +24 -5
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_select.py +4 -3
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_version.py +2 -2
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_calculators/_cache.py +198 -17
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_calculators/_hashstats.py +8 -1
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_calculators/_pixelstats.py +7 -3
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_calculators/_visualstats.py +64 -10
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_compute_ratios.py +12 -4
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_compute_stats.py +2 -7
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_rank.py +52 -35
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_crops.py +16 -10
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_invalidates.py +10 -5
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_merge.py +94 -12
- dataeval-1.1.2/src/dataeval/data/_relabel.py +397 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_split.py +3 -11
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_torchvision.py +4 -4
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_tracks.py +41 -9
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_view.py +35 -10
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/models/_backends.py +22 -12
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/models/_predictors.py +4 -4
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/protocols.py +138 -41
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/quality/_duplicates.py +12 -9
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/quality/_outliers.py +8 -5
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/utils/_internal.py +32 -30
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/utils/_validate.py +6 -7
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/utils/preprocessing.py +36 -12
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/utils/thresholds.py +18 -0
- dataeval-1.1.0rc6/src/dataeval/data/_relabel.py +0 -209
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/.gitignore +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/LICENSE +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_embeddings.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_experimental.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_helpers.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_log.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_aggregate.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_deprecated.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_encoding.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_entry_legacy.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_filters.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_input.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_links.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_loading.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_serialize.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_store.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_accumulator.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_base.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_block.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_classification.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_data.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_dataset.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_detection.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_factors.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_frames.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_gather.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_instances.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_layout.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_ordering.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_propagation.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_reporting.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_source_index.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_metadata/_structurers/_tracking.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/_ontology.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/bias/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/bias/_balance.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/bias/_diversity.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/bias/_parity.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/config.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_ber.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_bin.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_calculators/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_calculators/_base.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_calculators/_dimensionstats.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_calculators/_register.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_calculators/_registry.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_clusterer.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_completeness.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_coverage.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_divergence.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_diversity.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_fast_hdbscan/_cluster_trees.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_fast_hdbscan/_disjoint_set.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_fast_hdbscan/_mst.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_feature_distance.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_hash.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_label_alignment.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_label_coverage.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_label_errors.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_label_parity.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_label_reconciliation.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_label_stats.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_metadata_insights.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_mst.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_mutual_info.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_nullmodel.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_ontology_validation.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_parity.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_track_stats.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/core/_uap.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_classbalance.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_classfilter.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_crop.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_geometry.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_indices.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_limit.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_resize.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_reverse.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_selectchannels.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_shuffle.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/data/_unzip.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/exceptions.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/extractors/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/extractors/_bovw.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/extractors/_flatten.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/extractors/_geometry.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/extractors/_onnx.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/extractors/_scores.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/extractors/_torch.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/extractors/_uncertainty.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/flags.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/models/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/models/_input.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/models/_metadata.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/performance/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/performance/_aggregator.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/performance/_output.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/performance/_sufficiency.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/performance/schedules.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/py.typed +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/quality/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/quality/_shared.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/scope/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/scope/_coverage.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/scope/_prioritize.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/scope/_representation.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/selection/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_drift/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_drift/_base.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_drift/_chunk.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_drift/_domain_classifier.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_drift/_kneighbors.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_drift/_mmd.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_drift/_reconstruction.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_drift/_univariate.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_drift/_wasserstein.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_ood/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_ood/_base.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_ood/_domain_classifier.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_ood/_kneighbors.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_ood/_reconstruction.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_shared/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_shared/_domain_classifier.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_shared/_kneighbors.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/_shared/_reconstruction.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/shift/update_strategies.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/types/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/types/_array.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/types/_config.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/types/_evaluator.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/types/_execution.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/types/_factors.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/types/_index.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/types/_ontology.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/types/_output.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/types/_schema.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/types/_track.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/utils/__init__.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/utils/data.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/utils/losses.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/utils/models.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/utils/onnx.py +0 -0
- {dataeval-1.1.0rc6 → dataeval-1.1.2}/src/dataeval/utils/training.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: dataeval
|
|
3
|
-
Version: 1.1.
|
|
3
|
+
Version: 1.1.2
|
|
4
4
|
Summary: DataEval provides a simple interface to characterize image data and its impact on model performance across classification and object-detection tasks
|
|
5
5
|
Project-URL: Homepage, https://dataeval.ai/
|
|
6
6
|
Project-URL: Repository, https://github.com/aria-ml/dataeval/
|
|
@@ -35,23 +35,26 @@ Requires-Dist: xxhash>=3.4
|
|
|
35
35
|
Provides-Extra: cpu
|
|
36
36
|
Requires-Dist: torch>=2.2.0; extra == 'cpu'
|
|
37
37
|
Requires-Dist: torchvision>=0.17.0; extra == 'cpu'
|
|
38
|
-
Provides-Extra:
|
|
39
|
-
Requires-Dist: torch>=2.2.0; extra == '
|
|
40
|
-
Requires-Dist: torchvision>=0.17.0; extra == '
|
|
41
|
-
Provides-Extra:
|
|
42
|
-
Requires-Dist: torch>=2.2.0; extra == '
|
|
43
|
-
Requires-Dist: torchvision>=0.17.0; extra == '
|
|
38
|
+
Provides-Extra: cu126
|
|
39
|
+
Requires-Dist: torch>=2.2.0; extra == 'cu126'
|
|
40
|
+
Requires-Dist: torchvision>=0.17.0; extra == 'cu126'
|
|
41
|
+
Provides-Extra: cu130
|
|
42
|
+
Requires-Dist: torch>=2.2.0; extra == 'cu130'
|
|
43
|
+
Requires-Dist: torchvision>=0.17.0; extra == 'cu130'
|
|
44
44
|
Provides-Extra: litert
|
|
45
45
|
Requires-Dist: ai-edge-litert>=2.0; (python_version <= '3.14') and extra == 'litert'
|
|
46
46
|
Provides-Extra: onnx
|
|
47
47
|
Requires-Dist: onnx>=1.14.0; extra == 'onnx'
|
|
48
|
-
Requires-Dist: onnxruntime
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
Requires-Dist:
|
|
52
|
-
Requires-Dist: onnxruntime-gpu<1.
|
|
53
|
-
Requires-Dist: onnxruntime-gpu<1.27,>=1.
|
|
54
|
-
|
|
48
|
+
Requires-Dist: onnxruntime>=1.17; extra == 'onnx'
|
|
49
|
+
Provides-Extra: onnx-cu126
|
|
50
|
+
Requires-Dist: onnx>=1.14.0; extra == 'onnx-cu126'
|
|
51
|
+
Requires-Dist: onnxruntime-gpu<1.24,>=1.17; (python_version == '3.10' and extra != 'onnx-cu130') and extra == 'onnx-cu126'
|
|
52
|
+
Requires-Dist: onnxruntime-gpu<1.27,>=1.17; (python_version >= '3.11' and extra != 'onnx-cu130') and extra == 'onnx-cu126'
|
|
53
|
+
Requires-Dist: onnxruntime-gpu<1.27,>=1.24; (python_version >= '3.14' and extra != 'onnx-cu130') and extra == 'onnx-cu126'
|
|
54
|
+
Provides-Extra: onnx-cu130
|
|
55
|
+
Requires-Dist: onnx>=1.14.0; extra == 'onnx-cu130'
|
|
56
|
+
Requires-Dist: onnxruntime-gpu>=1.27; (python_version >= '3.11' and extra != 'onnx-cu126') and extra == 'onnx-cu130'
|
|
57
|
+
Requires-Dist: onnxruntime>=1.17; (python_version == '3.10' and extra != 'onnx-cu126') and extra == 'onnx-cu130'
|
|
55
58
|
Provides-Extra: ontology
|
|
56
59
|
Requires-Dist: rdflib>=7.0; extra == 'ontology'
|
|
57
60
|
Provides-Extra: opencv
|
|
@@ -143,8 +146,8 @@ variant, install `torch` from that variant's wheel index **first**, then install
|
|
|
143
146
|
DataEval — it accepts the build already present in the environment:
|
|
144
147
|
|
|
145
148
|
```bash
|
|
146
|
-
# 1. Pick your PyTorch build (cpu /
|
|
147
|
-
pip install torch --index-url https://download.pytorch.org/whl/
|
|
149
|
+
# 1. Pick your PyTorch build (cpu / cu126 / cu130)
|
|
150
|
+
pip install torch --index-url https://download.pytorch.org/whl/cu130
|
|
148
151
|
|
|
149
152
|
# 2. Install DataEval
|
|
150
153
|
pip install dataeval
|
|
@@ -159,7 +162,7 @@ pip install dataeval
|
|
|
159
162
|
> `pip install dataeval --extra-index-url https://download.pytorch.org/whl/cpu`
|
|
160
163
|
> does work, because the CPU index tracks the latest release.)
|
|
161
164
|
>
|
|
162
|
-
> **The `cpu` / `
|
|
165
|
+
> **The `cpu` / `cu126` / `cu130` extras do not select a PyTorch variant under
|
|
163
166
|
> pip.** All three declare the same requirements (`torch`, `torchvision`); what
|
|
164
167
|
> distinguishes them is `[tool.uv.sources]`, which routes those packages to the
|
|
165
168
|
> right wheel index. That is project metadata applied by uv when resolving **from
|
|
@@ -175,7 +178,7 @@ transform across a dataset view. If you want that class, install torchvision
|
|
|
175
178
|
yourself, from the **same index as your torch build**:
|
|
176
179
|
|
|
177
180
|
```bash
|
|
178
|
-
pip install torchvision --index-url https://download.pytorch.org/whl/
|
|
181
|
+
pip install torchvision --index-url https://download.pytorch.org/whl/cu130
|
|
179
182
|
```
|
|
180
183
|
|
|
181
184
|
> **Do not mix indexes.** A `torchvision` from PyPI alongside a torch installed
|
|
@@ -189,7 +192,7 @@ pip install torchvision --index-url https://download.pytorch.org/whl/cu128
|
|
|
189
192
|
### **Installing with uv**
|
|
190
193
|
|
|
191
194
|
```bash
|
|
192
|
-
uv pip install dataeval --torch-backend cpu # or
|
|
195
|
+
uv pip install dataeval --torch-backend cpu # or cu126 / cu130 / auto
|
|
193
196
|
```
|
|
194
197
|
|
|
195
198
|
### **Installing with conda**
|
|
@@ -83,8 +83,8 @@ variant, install `torch` from that variant's wheel index **first**, then install
|
|
|
83
83
|
DataEval — it accepts the build already present in the environment:
|
|
84
84
|
|
|
85
85
|
```bash
|
|
86
|
-
# 1. Pick your PyTorch build (cpu /
|
|
87
|
-
pip install torch --index-url https://download.pytorch.org/whl/
|
|
86
|
+
# 1. Pick your PyTorch build (cpu / cu126 / cu130)
|
|
87
|
+
pip install torch --index-url https://download.pytorch.org/whl/cu130
|
|
88
88
|
|
|
89
89
|
# 2. Install DataEval
|
|
90
90
|
pip install dataeval
|
|
@@ -99,7 +99,7 @@ pip install dataeval
|
|
|
99
99
|
> `pip install dataeval --extra-index-url https://download.pytorch.org/whl/cpu`
|
|
100
100
|
> does work, because the CPU index tracks the latest release.)
|
|
101
101
|
>
|
|
102
|
-
> **The `cpu` / `
|
|
102
|
+
> **The `cpu` / `cu126` / `cu130` extras do not select a PyTorch variant under
|
|
103
103
|
> pip.** All three declare the same requirements (`torch`, `torchvision`); what
|
|
104
104
|
> distinguishes them is `[tool.uv.sources]`, which routes those packages to the
|
|
105
105
|
> right wheel index. That is project metadata applied by uv when resolving **from
|
|
@@ -115,7 +115,7 @@ transform across a dataset view. If you want that class, install torchvision
|
|
|
115
115
|
yourself, from the **same index as your torch build**:
|
|
116
116
|
|
|
117
117
|
```bash
|
|
118
|
-
pip install torchvision --index-url https://download.pytorch.org/whl/
|
|
118
|
+
pip install torchvision --index-url https://download.pytorch.org/whl/cu130
|
|
119
119
|
```
|
|
120
120
|
|
|
121
121
|
> **Do not mix indexes.** A `torchvision` from PyPI alongside a torch installed
|
|
@@ -129,7 +129,7 @@ pip install torchvision --index-url https://download.pytorch.org/whl/cu128
|
|
|
129
129
|
### **Installing with uv**
|
|
130
130
|
|
|
131
131
|
```bash
|
|
132
|
-
uv pip install dataeval --torch-backend cpu # or
|
|
132
|
+
uv pip install dataeval --torch-backend cpu # or cu126 / cu130 / auto
|
|
133
133
|
```
|
|
134
134
|
|
|
135
135
|
### **Installing with conda**
|
|
@@ -47,20 +47,24 @@ dependencies = [
|
|
|
47
47
|
|
|
48
48
|
[project.optional-dependencies]
|
|
49
49
|
cpu = ["torch>=2.2.0", "torchvision>=0.17.0"]
|
|
50
|
-
|
|
51
|
-
|
|
50
|
+
cu126 = ["torch>=2.2.0", "torchvision>=0.17.0"]
|
|
51
|
+
cu130 = ["torch>=2.2.0", "torchvision>=0.17.0"]
|
|
52
52
|
litert = ["ai-edge-litert>=2.0; python_version <= '3.14'"]
|
|
53
53
|
opencv = ["opencv-python-headless>=4.8.0"]
|
|
54
54
|
onnx = [
|
|
55
55
|
"onnx>=1.14.0",
|
|
56
|
-
"onnxruntime>=1.
|
|
57
|
-
"onnxruntime>=1.15.0; python_version >= '3.11'",
|
|
56
|
+
"onnxruntime>=1.17",
|
|
58
57
|
]
|
|
59
|
-
onnx-
|
|
58
|
+
onnx-cu126 = [
|
|
60
59
|
"onnx>=1.14.0",
|
|
61
|
-
"onnxruntime-gpu>=1.
|
|
62
|
-
"onnxruntime-gpu>=1.
|
|
63
|
-
"onnxruntime-gpu>=1.24,<1.27; python_version >= '3.14'", # 1.24+ for cp314 wheels
|
|
60
|
+
"onnxruntime-gpu>=1.17,<1.24; python_version == '3.10' and extra != 'onnx-cu130'",
|
|
61
|
+
"onnxruntime-gpu>=1.17,<1.27; python_version >= '3.11' and extra != 'onnx-cu130'", # 1.27+ requires CUDA 13.0
|
|
62
|
+
"onnxruntime-gpu>=1.24,<1.27; python_version >= '3.14' and extra != 'onnx-cu130'", # 1.24+ for cp314 wheels
|
|
63
|
+
]
|
|
64
|
+
onnx-cu130 = [
|
|
65
|
+
"onnx>=1.14.0",
|
|
66
|
+
"onnxruntime>=1.17; python_version == '3.10' and extra != 'onnx-cu126'", # CUDA 13.0 is not supported on Python 3.10
|
|
67
|
+
"onnxruntime-gpu>=1.27; python_version >= '3.11' and extra != 'onnx-cu126'",
|
|
64
68
|
]
|
|
65
69
|
ontology = ["rdflib>=7.0"]
|
|
66
70
|
|
|
@@ -106,12 +110,13 @@ docsync = [
|
|
|
106
110
|
test = [
|
|
107
111
|
"coverage[toml]>=7.6",
|
|
108
112
|
"filelock>=3.20.3",
|
|
109
|
-
"onnx>=1.14.0",
|
|
110
|
-
"onnxscript>=0.6.0",
|
|
111
113
|
"pytest>=8.3",
|
|
112
114
|
"pytest-cov>=6.1",
|
|
113
115
|
"pytest-xdist>=3.6.1",
|
|
114
|
-
|
|
116
|
+
]
|
|
117
|
+
test-onnx = [
|
|
118
|
+
{ include-group = "test" },
|
|
119
|
+
"onnxscript>=0.6.0",
|
|
115
120
|
]
|
|
116
121
|
verify = [
|
|
117
122
|
"pytest>=8.3",
|
|
@@ -168,7 +173,7 @@ security = [ # keep in sync with [tool.uv.constraint-dependencies]
|
|
|
168
173
|
dev = [
|
|
169
174
|
{ include-group = "base" },
|
|
170
175
|
{ include-group = "lint" },
|
|
171
|
-
{ include-group = "test" },
|
|
176
|
+
{ include-group = "test-onnx" },
|
|
172
177
|
{ include-group = "type" },
|
|
173
178
|
{ include-group = "docs" },
|
|
174
179
|
"nox>=2025.5.1",
|
|
@@ -181,8 +186,13 @@ dev = [
|
|
|
181
186
|
conflicts = [
|
|
182
187
|
[
|
|
183
188
|
{ extra = "cpu" },
|
|
184
|
-
{ extra = "
|
|
185
|
-
{ extra = "
|
|
189
|
+
{ extra = "cu126" },
|
|
190
|
+
{ extra = "cu130" },
|
|
191
|
+
],
|
|
192
|
+
[
|
|
193
|
+
{ extra = "onnx" },
|
|
194
|
+
{ extra = "onnx-cu126" },
|
|
195
|
+
{ extra = "onnx-cu130" },
|
|
186
196
|
],
|
|
187
197
|
]
|
|
188
198
|
constraint-dependencies = [
|
|
@@ -208,25 +218,25 @@ url = "https://download.pytorch.org/whl/cpu"
|
|
|
208
218
|
explicit = true
|
|
209
219
|
|
|
210
220
|
[[tool.uv.index]]
|
|
211
|
-
name = "pytorch-
|
|
212
|
-
url = "https://download.pytorch.org/whl/
|
|
221
|
+
name = "pytorch-cu126"
|
|
222
|
+
url = "https://download.pytorch.org/whl/cu126"
|
|
213
223
|
explicit = true
|
|
214
224
|
|
|
215
225
|
[[tool.uv.index]]
|
|
216
|
-
name = "pytorch-
|
|
217
|
-
url = "https://download.pytorch.org/whl/
|
|
226
|
+
name = "pytorch-cu130"
|
|
227
|
+
url = "https://download.pytorch.org/whl/cu130"
|
|
218
228
|
explicit = true
|
|
219
229
|
|
|
220
230
|
[tool.uv.sources]
|
|
221
231
|
torch = [
|
|
222
232
|
{ index = "pytorch-cpu", extra = "cpu" },
|
|
223
|
-
{ index = "pytorch-
|
|
224
|
-
{ index = "pytorch-
|
|
233
|
+
{ index = "pytorch-cu126", extra = "cu126" },
|
|
234
|
+
{ index = "pytorch-cu130", extra = "cu130" },
|
|
225
235
|
]
|
|
226
236
|
torchvision = [
|
|
227
237
|
{ index = "pytorch-cpu", extra = "cpu" },
|
|
228
|
-
{ index = "pytorch-
|
|
229
|
-
{ index = "pytorch-
|
|
238
|
+
{ index = "pytorch-cu126", extra = "cu126" },
|
|
239
|
+
{ index = "pytorch-cu130", extra = "cu130" },
|
|
230
240
|
]
|
|
231
241
|
|
|
232
242
|
[tool.uv.extra-build-dependencies]
|
|
@@ -242,6 +252,7 @@ priority = "supplemental"
|
|
|
242
252
|
|
|
243
253
|
[tool.poetry.dependencies]
|
|
244
254
|
torch = { version = ">=2.2.0", source = "pytorch-cpu" }
|
|
255
|
+
torchvision = { version = ">=0.17.0", source = "pytorch-cpu" }
|
|
245
256
|
|
|
246
257
|
[tool.hatch.build.targets.sdist]
|
|
247
258
|
include = ["src/dataeval"]
|
|
@@ -264,12 +275,16 @@ vcs = "git"
|
|
|
264
275
|
style = "pep440"
|
|
265
276
|
pattern = "^v?(?P<base>\\d+\\.\\d+\\.\\d+)"
|
|
266
277
|
|
|
278
|
+
[tool.pyproject2conda]
|
|
279
|
+
# maite is only published on conda-forge; the defaults channel cannot resolve it.
|
|
280
|
+
channels = ["conda-forge"]
|
|
281
|
+
|
|
267
282
|
[tool.pyproject2conda.dependencies]
|
|
268
283
|
numpy = { skip = true, packages = "numpy>=1.24.2" }
|
|
269
284
|
scikit-learn = { skip = true, packages = "scikit-learn>=1.5.0" }
|
|
270
285
|
scipy = { skip = true, packages = "scipy>=1.10.0" }
|
|
271
286
|
torch = { pip = true } # PyTorch is no longer maintained on conda-forge
|
|
272
|
-
xxhash = { skip = true, packages = "python-xxhash>=3.
|
|
287
|
+
xxhash = { skip = true, packages = "python-xxhash>=3.4" }
|
|
273
288
|
|
|
274
289
|
[tool.pyright]
|
|
275
290
|
include = ["src", "tests", "verification", "docs/source/notebooks"]
|
|
@@ -312,7 +327,12 @@ omit = ["src/dataeval/_version.py"]
|
|
|
312
327
|
exclude_also = [
|
|
313
328
|
"raise NotImplementedError",
|
|
314
329
|
": \\.\\.\\.",
|
|
315
|
-
"if TYPE_CHECKING:"
|
|
330
|
+
"if TYPE_CHECKING:",
|
|
331
|
+
# Debug reprs carry no logic worth asserting on, and pinning their exact text in a
|
|
332
|
+
# test makes the string harder to improve than it is worth. coverage.py documents
|
|
333
|
+
# `def __repr__` as a canonical exclusion.
|
|
334
|
+
"def __repr__",
|
|
335
|
+
"def __str__",
|
|
316
336
|
]
|
|
317
337
|
include = ["*/src/dataeval/*"]
|
|
318
338
|
omit = [
|
|
@@ -345,7 +365,12 @@ extend-include = ["*.ipynb"]
|
|
|
345
365
|
select = ["F", "E", "W", "C90", "I", "N", "D", "UP", "YTT", "ANN", "S", "BLE", "B", "A",
|
|
346
366
|
"COM", "C4", "T10", "ISC", "ICN", "PYI", "PT", "Q", "RSE", "RET", "SLF", "SIM",
|
|
347
367
|
"TID252", "ARG", "FIX", "PD", "FLY", "NPY", "RUF027", "RUF100", "PERF"]
|
|
348
|
-
|
|
368
|
+
# ANN101/ANN102 were removed in Ruff 0.8, so ignoring them is inert and Ruff warns as much.
|
|
369
|
+
# They stay because the JATIC program-standards Ruff config requires them verbatim in
|
|
370
|
+
# lint.ignore, and verify_ruff_config.py matches against the parsed TOML -- a comment
|
|
371
|
+
# cannot satisfy it, and the only other accepted spelling is bare "ANN", which would
|
|
372
|
+
# disable every annotation rule.
|
|
373
|
+
ignore = ["ANN101", "ANN102", "ANN401", "C408", "C416", "COM812", "NPY002", "SLF001"]
|
|
349
374
|
fixable = ["ALL"]
|
|
350
375
|
unfixable = []
|
|
351
376
|
dummy-variable-rgx = "^(_+|(_+[a-zA-Z0-9_]*[a-zA-Z0-9]+?))$"
|
|
@@ -101,14 +101,38 @@ def split_by_dimensionality(
|
|
|
101
101
|
return kept, [name for name in arrays if name not in kept]
|
|
102
102
|
|
|
103
103
|
|
|
104
|
+
# The two suffixes binning appends, and the namespace they define between them. Named
|
|
105
|
+
# rather than spelled inline because :func:`is_companion_name` has to answer for the
|
|
106
|
+
# same characters these build with, and a suffix that drifted between the two would
|
|
107
|
+
# reopen exactly the collision that function exists to close.
|
|
108
|
+
BINNED_SUFFIX = "↕"
|
|
109
|
+
DIGITIZED_SUFFIX = "#"
|
|
110
|
+
COMPANION_SUFFIXES: tuple[str, ...] = (BINNED_SUFFIX, DIGITIZED_SUFFIX)
|
|
111
|
+
|
|
112
|
+
|
|
104
113
|
def binned(name: str) -> str:
|
|
105
114
|
"""Name of the companion column holding ``name``'s bin indices."""
|
|
106
|
-
return f"{name}
|
|
115
|
+
return f"{name}{BINNED_SUFFIX}"
|
|
107
116
|
|
|
108
117
|
|
|
109
118
|
def digitized(name: str) -> str:
|
|
110
119
|
"""Name of the companion column holding ``name``'s category ordinals."""
|
|
111
|
-
return f"{name}
|
|
120
|
+
return f"{name}{DIGITIZED_SUFFIX}"
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def is_companion_name(name: str) -> bool:
|
|
124
|
+
"""Whether ``name`` sits in the namespace binning writes its companion columns into.
|
|
125
|
+
|
|
126
|
+
Every reader that resolves a companion does it by construction — ``binned(col)`` and
|
|
127
|
+
``digitized(col)`` over the columns actually present — so a *factor* holding one of
|
|
128
|
+
those names is indistinguishable from the companion of its stem. A column named
|
|
129
|
+
``w#`` alongside a factor ``w`` makes ``Metadata._bin`` skip ``w`` as already binned,
|
|
130
|
+
makes ``_reset_bins`` and the serializer's ``_without_companions`` drop the caller's
|
|
131
|
+
values as derived, and leaves ``factor_names`` a name longer than ``factor_data`` is
|
|
132
|
+
wide. Reserving the namespace is what keeps all three honest, so this is consulted
|
|
133
|
+
wherever a factor is named — see ``safe_column_name`` in ``_structurers._reserved``.
|
|
134
|
+
"""
|
|
135
|
+
return name.endswith(COMPANION_SUFFIXES)
|
|
112
136
|
|
|
113
137
|
|
|
114
138
|
def to_col(name: str, info: FactorInfo, is_binned: bool = True) -> str:
|
|
@@ -61,6 +61,11 @@ def _item_values(md: "Metadata", factors: Mapping[str, Any], rows: int, key: str
|
|
|
61
61
|
``track_stats`` describes a single sequence and says nothing about which, so a dataset
|
|
62
62
|
holding exactly one item can supply the answer itself. A dataset holding several cannot:
|
|
63
63
|
track ids restart per sequence, so a bare id names a row in every one of them.
|
|
64
|
+
|
|
65
|
+
Which item a value belongs to has to be *said*, whether the caller attaches every
|
|
66
|
+
sequence at once or one per call. Repeated calls fold into one column rather than
|
|
67
|
+
colliding — see ``Metadata._merge_keyed`` — but each still has to name the item its
|
|
68
|
+
keys are scoped to.
|
|
64
69
|
"""
|
|
65
70
|
if _ITEM in factors:
|
|
66
71
|
return np.asarray(factors[_ITEM], dtype=np.intp).reshape(-1)
|
|
@@ -70,8 +75,10 @@ def _item_values(md: "Metadata", factors: Mapping[str, Any], rows: int, key: str
|
|
|
70
75
|
raise ValueError(
|
|
71
76
|
f"key={key!r} matches on (item_index, {key}), and {key} restarts per item, so values "
|
|
72
77
|
f"for a dataset with {len(items)} items have to say which item each belongs to. Add an "
|
|
73
|
-
f"'item_index' entry to the factors
|
|
74
|
-
"describes one sequence at a time
|
|
78
|
+
f"'item_index' entry to the factors — one entry per value, naming the item that value's "
|
|
79
|
+
f"{key} is scoped to. track_stats describes one sequence at a time, so attaching a "
|
|
80
|
+
"dataset's worth of them means saying which sequence each result came from, whether "
|
|
81
|
+
"they go in one call or one call per sequence.",
|
|
75
82
|
)
|
|
76
83
|
|
|
77
84
|
|
|
@@ -80,13 +87,17 @@ def resolve_keyed(
|
|
|
80
87
|
factors: Mapping[str, Any],
|
|
81
88
|
level: FactorLevel,
|
|
82
89
|
key: str,
|
|
83
|
-
) -> list[tuple[str, FactorLevel, pl.Series]]:
|
|
90
|
+
) -> tuple[list[tuple[str, FactorLevel, pl.Series]], NDArray[np.bool_]]:
|
|
84
91
|
"""Place each factor on the rows whose ``(item_index, key)`` its values name.
|
|
85
92
|
|
|
86
93
|
A row the incoming values do not name is null rather than absent, so the column still
|
|
87
94
|
has one entry per row at ``level`` and every downstream reader — binning, projection,
|
|
88
95
|
the flat frame — sees the shape it expects.
|
|
89
96
|
|
|
97
|
+
Which rows *were* named is returned alongside, because it is the difference between a
|
|
98
|
+
write that leaves the rest of the column alone and one that blanks it. Every factor in
|
|
99
|
+
a call is placed by the same keys, so one mask covers them all.
|
|
100
|
+
|
|
90
101
|
Parameters
|
|
91
102
|
----------
|
|
92
103
|
md : Metadata
|
|
@@ -103,6 +114,8 @@ def resolve_keyed(
|
|
|
103
114
|
-------
|
|
104
115
|
list[tuple[str, str, pl.Series]]
|
|
105
116
|
One entry per remaining factor, already in the level's row order.
|
|
117
|
+
NDArray[np.bool_]
|
|
118
|
+
One flag per row at ``level``, True where the incoming keys named it.
|
|
106
119
|
|
|
107
120
|
Raises
|
|
108
121
|
------
|
|
@@ -148,4 +161,7 @@ def resolve_keyed(
|
|
|
148
161
|
source = {pair: position for position, pair in enumerate(incoming)}
|
|
149
162
|
wanted = zip(frame["item_index"].to_list(), frame[key].to_list(), strict=True)
|
|
150
163
|
positions = np.fromiter((source.get(pair, -1) for pair in wanted), dtype=np.intp, count=frame.height)
|
|
151
|
-
|
|
164
|
+
placed: list[tuple[str, FactorLevel, pl.Series]] = [
|
|
165
|
+
(name, level, gather_nulling(name, values, positions)) for name, values in payload.items()
|
|
166
|
+
]
|
|
167
|
+
return placed, positions >= 0
|
|
@@ -66,6 +66,7 @@ from dataeval.core._bin import (
|
|
|
66
66
|
level_budget,
|
|
67
67
|
)
|
|
68
68
|
from dataeval.core._compute_stats import StatsResult
|
|
69
|
+
from dataeval.core._track_stats import TrackStatsResult
|
|
69
70
|
from dataeval.exceptions import NotFittedError, ShapeMismatchError
|
|
70
71
|
from dataeval.protocols import (
|
|
71
72
|
AnnotatedDataset,
|
|
@@ -1476,7 +1477,7 @@ class Metadata(DeprecatedMetadataAPI, Array, FeatureExtractor):
|
|
|
1476
1477
|
-------
|
|
1477
1478
|
Metadata
|
|
1478
1479
|
A copy whose :attr:`view` is ``level``, sharing this instance's structuring
|
|
1479
|
-
and binning work.
|
|
1480
|
+
and binning work.
|
|
1480
1481
|
|
|
1481
1482
|
Raises
|
|
1482
1483
|
------
|
|
@@ -1490,6 +1491,14 @@ class Metadata(DeprecatedMetadataAPI, Array, FeatureExtractor):
|
|
|
1490
1491
|
the metadata is being handed to an evaluator, so that two evaluators can read
|
|
1491
1492
|
two levels of the same dataset at once.
|
|
1492
1493
|
|
|
1494
|
+
The original reports every value it reported before, but it is not left alone:
|
|
1495
|
+
structuring and binning run on it here if they have not run already, so that the
|
|
1496
|
+
copy shares that work instead of repeating it on a store of its own. Binning adds
|
|
1497
|
+
companion columns and bins each factor at its own level, so nothing readable
|
|
1498
|
+
moves — what moves is *when*. A warning a factor would have raised at the copy's
|
|
1499
|
+
first factor access is raised at this call instead, and a binning configuration
|
|
1500
|
+
that cannot be applied fails here rather than there.
|
|
1501
|
+
|
|
1493
1502
|
Examples
|
|
1494
1503
|
--------
|
|
1495
1504
|
>>> metadata = Metadata(dataset)
|
|
@@ -1501,6 +1510,7 @@ class Metadata(DeprecatedMetadataAPI, Array, FeatureExtractor):
|
|
|
1501
1510
|
50
|
|
1502
1511
|
"""
|
|
1503
1512
|
self._structure()
|
|
1513
|
+
self._bin()
|
|
1504
1514
|
resolved = self._resolve_level(level)
|
|
1505
1515
|
|
|
1506
1516
|
view = copy.copy(self)
|
|
@@ -2059,10 +2069,25 @@ class Metadata(DeprecatedMetadataAPI, Array, FeatureExtractor):
|
|
|
2059
2069
|
-------
|
|
2060
2070
|
pl.DataFrame
|
|
2061
2071
|
DataFrame with columns for level, item_index, target_index, class_label,
|
|
2062
|
-
|
|
2072
|
+
score, bounding boxes (when applicable), a ``level`` tag naming the
|
|
2063
2073
|
level each row belongs to, that level's own key columns, and all
|
|
2064
2074
|
processed metadata factors.
|
|
2065
2075
|
|
|
2076
|
+
``score`` holds whatever layout the dataset's targets carried: one
|
|
2077
|
+
confidence per labelled row, or a per-class array as wide as the
|
|
2078
|
+
vocabulary that produced it.
|
|
2079
|
+
|
|
2080
|
+
.. note::
|
|
2081
|
+
v1.2 reads ``score`` down to one ``Float32`` per row — the row's
|
|
2082
|
+
confidence in its **own** class — and spells an unreadable one as
|
|
2083
|
+
null. A per-class array's width is a property of the dataset's class
|
|
2084
|
+
count, which is why two datasets with different vocabularies cannot
|
|
2085
|
+
be stacked into one frame today. Code recovering per-class
|
|
2086
|
+
probabilities from this column should read them from the target
|
|
2087
|
+
instead. :class:`~dataeval.data.Relabel` takes
|
|
2088
|
+
``reduce_detection_scores`` to adopt the new column shape now, or to
|
|
2089
|
+
keep this one through v1.2.
|
|
2090
|
+
|
|
2066
2091
|
See Also
|
|
2067
2092
|
--------
|
|
2068
2093
|
:meth:`~dataeval.Metadata.rows_at` : Filter to any level
|
|
@@ -2086,6 +2111,7 @@ class Metadata(DeprecatedMetadataAPI, Array, FeatureExtractor):
|
|
|
2086
2111
|
still — neither goes through this.
|
|
2087
2112
|
"""
|
|
2088
2113
|
self._structure()
|
|
2114
|
+
self._bin()
|
|
2089
2115
|
if self._flat is None:
|
|
2090
2116
|
self._flat = self._store.flat()
|
|
2091
2117
|
return self._flat
|
|
@@ -2733,6 +2759,7 @@ class Metadata(DeprecatedMetadataAPI, Array, FeatureExtractor):
|
|
|
2733
2759
|
50
|
|
2734
2760
|
"""
|
|
2735
2761
|
self._structure()
|
|
2762
|
+
self._bin()
|
|
2736
2763
|
return self._store.resolve(self._resolve_level(level))
|
|
2737
2764
|
|
|
2738
2765
|
def _empty_projection(self, dtype: Any) -> NDArray[Any]:
|
|
@@ -3462,17 +3489,63 @@ class Metadata(DeprecatedMetadataAPI, Array, FeatureExtractor):
|
|
|
3462
3489
|
self._announce_derived_encodings(factor_info)
|
|
3463
3490
|
self._announce_fit(factor_info)
|
|
3464
3491
|
|
|
3492
|
+
def _merge_keyed(
|
|
3493
|
+
self,
|
|
3494
|
+
name: str,
|
|
3495
|
+
level: FactorLevel,
|
|
3496
|
+
values: Any,
|
|
3497
|
+
named: NDArray[np.bool_],
|
|
3498
|
+
overwrite: bool,
|
|
3499
|
+
) -> tuple[str, pl.Series] | None:
|
|
3500
|
+
"""Fold a keyed write into a column of the same name already held at that level.
|
|
3501
|
+
|
|
3502
|
+
A keyed write names *rows*. Reaching rows that no earlier write reached is not a
|
|
3503
|
+
name collision even though the column exists — it is the rest of the same column
|
|
3504
|
+
arriving. Attaching per-sequence results one item at a time has exactly that
|
|
3505
|
+
shape, and :func:`~dataeval.core.track_stats` describes one sequence at a time, so
|
|
3506
|
+
it is the shape a caller naturally writes. Treating it as a collision instead
|
|
3507
|
+
leaves two half-null columns under two names and says nothing about it.
|
|
3508
|
+
|
|
3509
|
+
Returns
|
|
3510
|
+
-------
|
|
3511
|
+
tuple[str, pl.Series] or None
|
|
3512
|
+
The column to write and its merged values, or None when there is nothing to
|
|
3513
|
+
fold into or the write collides for real.
|
|
3514
|
+
|
|
3515
|
+
Notes
|
|
3516
|
+
-----
|
|
3517
|
+
None comes back in two cases. The level holds no such factor, so this is a first
|
|
3518
|
+
write and there is nothing to merge; or a row this write names already holds a
|
|
3519
|
+
value while `overwrite` is False, which is two values for one row and so a real
|
|
3520
|
+
collision — left to :meth:`_resolve_factor_name` to rename, like any other.
|
|
3521
|
+
|
|
3522
|
+
Under ``overwrite=True`` the named rows are replaced and the rest are kept, rather
|
|
3523
|
+
than the whole column being replaced. Rows this write does not name are not rows
|
|
3524
|
+
it says anything about.
|
|
3525
|
+
"""
|
|
3526
|
+
safe = safe_column_name(name)
|
|
3527
|
+
if safe not in self._factors_by_level.get(level, ()):
|
|
3528
|
+
return None
|
|
3529
|
+
existing = self._store.frame(level)[safe]
|
|
3530
|
+
written = pl.Series(named)
|
|
3531
|
+
if not overwrite and existing.filter(written).is_not_null().any():
|
|
3532
|
+
return None
|
|
3533
|
+
return safe, to_series(safe, values).zip_with(written, existing)
|
|
3534
|
+
|
|
3465
3535
|
def _resolve_factor_name(self, name: str, taken: set[str], overwrite: bool, append_string: str) -> str:
|
|
3466
3536
|
"""Pick the dataframe column a new factor should be written to.
|
|
3467
3537
|
|
|
3468
3538
|
Reserved columns are load-bearing — ``level`` drives every level filter — so a
|
|
3469
|
-
colliding factor is renamed rather than allowed to overwrite one.
|
|
3539
|
+
colliding factor is renamed rather than allowed to overwrite one. So is one named
|
|
3540
|
+
into the namespace binning writes its companions into, which ``taken`` cannot
|
|
3541
|
+
speak for: it holds the columns present *now*, and a companion this factor would
|
|
3542
|
+
be mistaken for may not have been written yet.
|
|
3470
3543
|
"""
|
|
3471
3544
|
safe = safe_column_name(name)
|
|
3472
3545
|
if safe != name:
|
|
3473
3546
|
_logger.warning(
|
|
3474
|
-
f"The factor name '{name}' collides with a
|
|
3475
|
-
f"stored as '{safe}' instead.",
|
|
3547
|
+
f"The factor name '{name}' collides with a column name DataEval reserves and has "
|
|
3548
|
+
f"been stored as '{safe}' instead.",
|
|
3476
3549
|
)
|
|
3477
3550
|
|
|
3478
3551
|
if safe not in taken or overwrite:
|
|
@@ -3658,7 +3731,7 @@ class Metadata(DeprecatedMetadataAPI, Array, FeatureExtractor):
|
|
|
3658
3731
|
|
|
3659
3732
|
def add_factors(
|
|
3660
3733
|
self,
|
|
3661
|
-
factors: Mapping[str, Array1D[Any]] | StatsResult,
|
|
3734
|
+
factors: Mapping[str, Array1D[Any]] | StatsResult | TrackStatsResult,
|
|
3662
3735
|
level: FactorLevel | Literal["auto", "target", "combined", "image"] = "auto",
|
|
3663
3736
|
overwrite: bool = False,
|
|
3664
3737
|
append_string: str = "_added",
|
|
@@ -3724,6 +3797,9 @@ class Metadata(DeprecatedMetadataAPI, Array, FeatureExtractor):
|
|
|
3724
3797
|
overwrite : bool, default False
|
|
3725
3798
|
Whether to overwrite factors of the same name already present in the metadata.
|
|
3726
3799
|
When False, a colliding factor is stored under a new name instead (see `append_string`).
|
|
3800
|
+
|
|
3801
|
+
Under `key` a collision is decided per row rather than per name, since a keyed
|
|
3802
|
+
write names rows: see the `key` description below.
|
|
3727
3803
|
append_string : str, default "_added"
|
|
3728
3804
|
Suffix appended to a factor name that collides with an existing column when
|
|
3729
3805
|
`overwrite` is False. If the suffixed name is also taken, an incrementing
|
|
@@ -3752,6 +3828,15 @@ class Metadata(DeprecatedMetadataAPI, Array, FeatureExtractor):
|
|
|
3752
3828
|
``track_ids``, and both that and the singular column name are accepted. A row
|
|
3753
3829
|
no incoming key names is null, so the column still has one value per row.
|
|
3754
3830
|
|
|
3831
|
+
Because a keyed write names rows, a second one adding a factor already present
|
|
3832
|
+
**folds into that column** rather than colliding with it: rows the new keys
|
|
3833
|
+
name take the new values, and rows they do not are left as they were. Attaching
|
|
3834
|
+
one sequence per call therefore builds a single column across the whole dataset,
|
|
3835
|
+
which is what ``track_stats`` invites, describing one sequence at a time. A name
|
|
3836
|
+
collision is reported only when a row that already holds a value is named again,
|
|
3837
|
+
and `overwrite` then decides it as it does anywhere else — replacing just the
|
|
3838
|
+
named rows rather than the whole column.
|
|
3839
|
+
|
|
3755
3840
|
Raises
|
|
3756
3841
|
------
|
|
3757
3842
|
ShapeMismatchError
|
|
@@ -3852,14 +3937,19 @@ class Metadata(DeprecatedMetadataAPI, Array, FeatureExtractor):
|
|
|
3852
3937
|
|
|
3853
3938
|
taken = set(self._store.columns)
|
|
3854
3939
|
resolved: list[_ResolvedFactor] = []
|
|
3940
|
+
named: NDArray[np.bool_] | None = None
|
|
3855
3941
|
if key is not None:
|
|
3856
3942
|
# _reject_unusable_key has already refused "auto" and "combined", the only
|
|
3857
3943
|
# spellings resolving to something other than a level, so this is one.
|
|
3858
|
-
placed, vacuous = resolve_keyed(self, kept, cast("FactorLevel", resolved_level), key), []
|
|
3944
|
+
(placed, named), vacuous = resolve_keyed(self, kept, cast("FactorLevel", resolved_level), key), []
|
|
3859
3945
|
else:
|
|
3860
3946
|
placed, vacuous = self._resolve_factor_levels(kept, resolved_level, source_index)
|
|
3861
3947
|
for name, factor_level, values in placed:
|
|
3862
|
-
|
|
3948
|
+
merged = None if named is None else self._merge_keyed(name, factor_level, values, named, overwrite)
|
|
3949
|
+
if merged is not None:
|
|
3950
|
+
col_name, values = merged
|
|
3951
|
+
else:
|
|
3952
|
+
col_name = self._resolve_factor_name(name, taken, overwrite, append_string)
|
|
3863
3953
|
taken.add(col_name)
|
|
3864
3954
|
# One value per entity at the factor's own level: descendant rows read them by
|
|
3865
3955
|
# the store's gather, so there is no expanded copy to build and no dtype to
|
|
@@ -4,7 +4,8 @@ A dataframe row carries two kinds of column. Factors are observations — anythi
|
|
|
4
4
|
dataset or the caller measured — and are binned, correlated and reported on. The
|
|
5
5
|
reserved columns are the row's own identity: the level it belongs to, the item it came
|
|
6
6
|
from, and where it sits within each of its parents. A factor whose name would collide
|
|
7
|
-
with one of them is renamed rather than allowed to overwrite it
|
|
7
|
+
with one of them is renamed rather than allowed to overwrite it, and so is one that
|
|
8
|
+
would be taken for a companion column binning writes — see :func:`safe_column_name`.
|
|
8
9
|
|
|
9
10
|
Sole producer of that layout: every structurer and
|
|
10
11
|
:meth:`~dataeval.Metadata.from_factors` builds its blocks through
|
|
@@ -20,6 +21,7 @@ from typing import Any
|
|
|
20
21
|
import numpy as np
|
|
21
22
|
from numpy.typing import NDArray
|
|
22
23
|
|
|
24
|
+
from dataeval._metadata._columns import is_companion_name
|
|
23
25
|
from dataeval.types import FactorLevel
|
|
24
26
|
|
|
25
27
|
# Columns the metadata dataframe has always carried. Retained verbatim because
|
|
@@ -149,7 +151,22 @@ def _as_column(values: Any) -> Sequence[Any] | NDArray[Any]:
|
|
|
149
151
|
|
|
150
152
|
|
|
151
153
|
def safe_column_name(name: str) -> str:
|
|
152
|
-
"""
|
|
154
|
+
"""Rename a factor that would be taken for a column DataEval owns.
|
|
155
|
+
|
|
156
|
+
Two namespaces are reserved, and a factor is moved out of either rather than allowed
|
|
157
|
+
to occupy it. :data:`RESERVED_COLUMNS` is the row's own identity, collided with head-on
|
|
158
|
+
and escaped by prefix. The companion namespace — anything ending in one of
|
|
159
|
+
``COMPANION_SUFFIXES`` — is the one binning writes into, so a name lands in it by its
|
|
160
|
+
*tail* and has to be escaped there; see ``is_companion_name`` in ``_metadata._columns``
|
|
161
|
+
for what mistaking the two costs.
|
|
162
|
+
|
|
163
|
+
Sole entry point for both: every factor name reaches a frame through here, whether it
|
|
164
|
+
came from a dataset's metadata dictionaries, from
|
|
165
|
+
:meth:`~dataeval.Metadata.from_factors`, or from
|
|
166
|
+
:meth:`~dataeval.Metadata.add_factors` and :meth:`~dataeval.Metadata.agg` by way of
|
|
167
|
+
``_resolve_factor_name``. Placed here rather than in each caller because the check has
|
|
168
|
+
to hold before anything is binned: a writer that resolves its name against the columns
|
|
169
|
+
currently present cannot see a companion binning has not written yet.
|
|
153
170
|
|
|
154
171
|
Parameters
|
|
155
172
|
----------
|
|
@@ -159,7 +176,9 @@ def safe_column_name(name: str) -> str:
|
|
|
159
176
|
Returns
|
|
160
177
|
-------
|
|
161
178
|
str
|
|
162
|
-
``name`` unchanged
|
|
163
|
-
|
|
179
|
+
``name`` unchanged; ``metadata_<name>`` when it is in :data:`RESERVED_COLUMNS`;
|
|
180
|
+
or ``<name>_metadata`` when it ends in a companion suffix.
|
|
164
181
|
"""
|
|
165
|
-
|
|
182
|
+
if name in RESERVED_COLUMNS:
|
|
183
|
+
return f"metadata_{name}"
|
|
184
|
+
return f"{name}_metadata" if is_companion_name(name) else name
|