datatunner 3.2.1__tar.gz → 3.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {datatunner-3.2.1/datatunner.egg-info → datatunner-3.2.2}/PKG-INFO +1 -1
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/orchestrator.py +14 -2
- {datatunner-3.2.1 → datatunner-3.2.2/datatunner.egg-info}/PKG-INFO +1 -1
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner.egg-info/SOURCES.txt +1 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/pyproject.toml +2 -2
- datatunner-3.2.2/tests/test_image_support.py +39 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/LICENSE +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/MANIFEST.in +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/README.md +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/domain/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/domain/dataset.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/domain/experiment.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/domain/generator.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/domain/metrics.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/evaluation/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/evaluation/performance/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/evaluation/performance/classification.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/evaluation/performance/regression.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/evaluation/quality/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/evaluation/quality/coverage.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/evaluation/quality/fidelity.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/evaluation/quality/statistical.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/exceptions.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/generators/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/generators/augmentation.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/generators/base.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/generators/ctgan.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/generators/smote.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/infrastructure/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/infrastructure/hardware.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/infrastructure/logging.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/infrastructure/persistence.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/infrastructure/seed.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/mixing/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/mixing/engine.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/mixing/strategies.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/mixing/validators.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/optimization/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/optimization/base.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/optimization/bayesian.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/optimization/grid.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/optimization/random.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/py.typed +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/reporting/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/reporting/exporters.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/reporting/plots.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/reporting/tables.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/training/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/training/checkpoint.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/training/environment.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner/training/runner.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner.egg-info/dependency_links.txt +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner.egg-info/requires.txt +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/datatunner.egg-info/top_level.txt +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/examples/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/examples/example_adult_ctgan.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/examples/example_adult_smote.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/examples/example_cifar_augmentation.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/examples/example_colab_breast_cancer.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/examples/example_colab_drybean_ctgan.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/examples/example_colab_mnist_augmentation.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/examples/example_colab_scatter_per_alpha.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/examples/example_comparison_smote_vs_ctgan.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/setup.cfg +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/tests/__init__.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/tests/test_aggregation.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/tests/test_domain.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/tests/test_generators.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/tests/test_infrastructure.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/tests/test_metrics_formatting.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/tests/test_mixing.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/tests/test_orchestrator_regressions.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/tests/test_seed_counter.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/tests/test_training_defaults.py +0 -0
- {datatunner-3.2.1 → datatunner-3.2.2}/tests/test_validators_coverage.py +0 -0
|
@@ -236,6 +236,18 @@ class DataTunner:
|
|
|
236
236
|
|
|
237
237
|
return report
|
|
238
238
|
|
|
239
|
+
@staticmethod
|
|
240
|
+
def _count_samples(data: Any) -> int:
|
|
241
|
+
"""Number of samples in a dataset, handling (images, labels) tuples.
|
|
242
|
+
|
|
243
|
+
Image pipelines pass real_data as a (X, y) tuple; ``len()`` on the
|
|
244
|
+
tuple returns the number of elements (2), not the number of samples.
|
|
245
|
+
Without this, n_synthetic = int(alpha * 2) ≈ 0 for every alpha.
|
|
246
|
+
"""
|
|
247
|
+
if isinstance(data, tuple) and len(data) == 2:
|
|
248
|
+
return len(data[0])
|
|
249
|
+
return len(data)
|
|
250
|
+
|
|
239
251
|
def _run_single_experiment(self, alpha: float, seed: int) -> ExperimentResult:
|
|
240
252
|
"""Execute ONE complete experiment: generate -> mix -> train -> evaluate.
|
|
241
253
|
|
|
@@ -245,7 +257,7 @@ class DataTunner:
|
|
|
245
257
|
|
|
246
258
|
with self.environment.isolate(seed, experiment_id):
|
|
247
259
|
# 1. Generate synthetic data
|
|
248
|
-
n_real =
|
|
260
|
+
n_real = self._count_samples(self._real_data)
|
|
249
261
|
n_synthetic = int(alpha * n_real)
|
|
250
262
|
|
|
251
263
|
# Bind the generator RNG to this trial's seed so repetitions
|
|
@@ -346,7 +358,7 @@ class DataTunner:
|
|
|
346
358
|
) -> List[Dict]:
|
|
347
359
|
"""Pre-evaluate generator by generating a sample and measuring fidelity."""
|
|
348
360
|
# Generate a fixed-size sample for quality assessment
|
|
349
|
-
sample_size = min(1000,
|
|
361
|
+
sample_size = min(1000, self._count_samples(real_data))
|
|
350
362
|
generator.set_seed(generator.spec.random_state)
|
|
351
363
|
synthetic_sample = generator.generate(sample_size)
|
|
352
364
|
|
|
@@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "datatunner"
|
|
7
|
-
# Bump on release: the current published version on PyPI is 3.
|
|
8
|
-
version = "3.2.
|
|
7
|
+
# Bump on release: the current published version on PyPI is 3.2.1.
|
|
8
|
+
version = "3.2.2"
|
|
9
9
|
description = "DataTunner: Scientific Platform for Optimal Artificial Data Proportion in Deep Learning"
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
license = "MIT"
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Regression tests for image pipeline support (X, y) tuples.
|
|
2
|
+
|
|
3
|
+
Bug fix (2026-08): `_run_single_experiment` computed n_synthetic with
|
|
4
|
+
``len(real_data)``. For image pipelines real_data is a (X, y) tuple, so
|
|
5
|
+
len() returned 2 → n_synthetic = int(alpha * 2) ≈ 0 for every alpha:
|
|
6
|
+
the ImageAugmentation pipeline generated (almost) no synthetic data.
|
|
7
|
+
|
|
8
|
+
Fix: DataTunner._count_samples() counts rows of data[0] for 2-tuples.
|
|
9
|
+
Used in _run_single_experiment and _evaluate_generator_quality.
|
|
10
|
+
|
|
11
|
+
Run where deps are installed (CI / Colab): pytest tests/test_image_support.py
|
|
12
|
+
"""
|
|
13
|
+
import pytest
|
|
14
|
+
|
|
15
|
+
from datatunner.orchestrator import DataTunner
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class TestCountSamples:
|
|
19
|
+
@pytest.mark.parametrize("n", [0, 1, 2, 10, 1000])
|
|
20
|
+
def test_tuple_counts_samples_not_elements(self, n):
|
|
21
|
+
X = [0] * n
|
|
22
|
+
y = [0] * n
|
|
23
|
+
assert DataTunner._count_samples((X, y)) == n
|
|
24
|
+
|
|
25
|
+
def test_nested_2d_like_image_array(self):
|
|
26
|
+
# X shaped (N, H, W): must count N, not dimensions
|
|
27
|
+
X = [[[0] * 4 for _ in range(4)] for _ in range(250)]
|
|
28
|
+
y = [0] * 250
|
|
29
|
+
assert DataTunner._count_samples((X, y)) == 250
|
|
30
|
+
|
|
31
|
+
def test_plain_list_counts_len(self):
|
|
32
|
+
assert DataTunner._count_samples([1, 2, 3, 4]) == 4
|
|
33
|
+
|
|
34
|
+
def test_single_element_list(self):
|
|
35
|
+
assert DataTunner._count_samples(["only"]) == 1
|
|
36
|
+
|
|
37
|
+
def test_non_tuple_pair_not_misclassified(self):
|
|
38
|
+
# A 2-element tuple is ambiguous; contract: (X, y) tuples mean images
|
|
39
|
+
assert DataTunner._count_samples((1, 2)) == 1 # len(data[0]) = 1
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|