dataflow-cv 2.0.0__tar.gz → 2.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/PKG-INFO +3 -4
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/README.md +1 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/__init__.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/filter.py +30 -58
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/log_templates.py +27 -29
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/partition.py +17 -46
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/sample.py +8 -23
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/split.py +8 -27
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/stats.py +23 -21
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/utils.py +12 -31
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/__init__.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/__init__.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/analyse.py +18 -34
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/convert.py +28 -15
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/evaluate.py +16 -4
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/utils.py +12 -10
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/visualize.py +15 -7
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/exceptions.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/main.py +14 -9
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/base.py +49 -74
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/coco_and_labelme.py +31 -30
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/labelme_and_yolo.py +30 -40
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/rle_converter.py +3 -10
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/utils.py +16 -42
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/yolo_and_coco.py +33 -44
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/base.py +6 -16
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/evaluator.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/log_templates.py +40 -14
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/metrics.py +28 -32
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/result.py +23 -26
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/utils.py +13 -32
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/__init__.py +8 -2
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/base.py +20 -48
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/coco_handler.py +48 -82
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/labelme_handler.py +51 -99
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/models.py +7 -21
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/utils.py +2 -6
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/yolo_handler.py +78 -143
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/util/logging.py +8 -28
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/__init__.py +1 -2
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/base.py +21 -60
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/coco_visualizer.py +4 -10
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/labelme_visualizer.py +2 -7
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/log_templates.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/utils.py +0 -1
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/yolo_visualizer.py +4 -12
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/PKG-INFO +3 -4
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/requires.txt +1 -3
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/pyproject.toml +10 -26
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/LICENSE +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/__init__.py +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/base.py +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/__init__.py +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/log_templates.py +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/__init__.py +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/util/__init__.py +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/SOURCES.txt +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/dependency_links.txt +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/entry_points.txt +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/not-zip-safe +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/top_level.txt +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dataflow-cv
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.1
|
|
4
4
|
Summary: A computer vision dataset processing library — analyse, convert, visualize, and evaluate annotations across YOLO, LabelMe, and COCO formats
|
|
5
5
|
Author: DataFlow-CV Team
|
|
6
6
|
License: MIT
|
|
@@ -33,9 +33,7 @@ Requires-Dist: pycocotools>=2.0.0; extra == "coco"
|
|
|
33
33
|
Provides-Extra: dev
|
|
34
34
|
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
35
35
|
Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
|
|
36
|
-
Requires-Dist:
|
|
37
|
-
Requires-Dist: isort>=5.12.0; extra == "dev"
|
|
38
|
-
Requires-Dist: flake8>=6.0.0; extra == "dev"
|
|
36
|
+
Requires-Dist: ruff>=0.16; extra == "dev"
|
|
39
37
|
Requires-Dist: mypy>=1.0.0; extra == "dev"
|
|
40
38
|
Dynamic: license-file
|
|
41
39
|
|
|
@@ -67,6 +65,7 @@ A computer vision dataset processing library — analyse, convert, visualize, an
|
|
|
67
65
|
| 🎨 **Visualize** | OpenCV rendering with color-coded classes, display & save modes | `dataflow-cv visualize yolo ...` |
|
|
68
66
|
| 📊 **Evaluate** | COCO mAP via pycocotools, single-threshold P/R/F1 per class | `dataflow-cv evaluate detection ...` |
|
|
69
67
|
| 💻 **CLI + API** | Click-based CLI with rich `--help`; Python API for pipelines | `from dataflow.convert import ...` |
|
|
68
|
+
| 🤖 **AI Skills** | Claude Code skill (`/dataflow:dataflow-cv`) for AI assistants — CLI/API reference & known gotchas | `claude plugin install dataflow@claude-skills` |
|
|
70
69
|
|
|
71
70
|
---
|
|
72
71
|
|
|
@@ -26,6 +26,7 @@ A computer vision dataset processing library — analyse, convert, visualize, an
|
|
|
26
26
|
| 🎨 **Visualize** | OpenCV rendering with color-coded classes, display & save modes | `dataflow-cv visualize yolo ...` |
|
|
27
27
|
| 📊 **Evaluate** | COCO mAP via pycocotools, single-threshold P/R/F1 per class | `dataflow-cv evaluate detection ...` |
|
|
28
28
|
| 💻 **CLI + API** | Click-based CLI with rich `--help`; Python API for pipelines | `from dataflow.convert import ...` |
|
|
29
|
+
| 🤖 **AI Skills** | Claude Code skill (`/dataflow:dataflow-cv`) for AI assistants — CLI/API reference & known gotchas | `claude plugin install dataflow@claude-skills` |
|
|
29
30
|
|
|
30
31
|
---
|
|
31
32
|
|
|
@@ -24,8 +24,7 @@ from .log_templates import (
|
|
|
24
24
|
format_filter_result,
|
|
25
25
|
)
|
|
26
26
|
from .utils import create_handler, detect_format, load_class_names
|
|
27
|
-
from dataflow.label.models import
|
|
28
|
-
ObjectAnnotation)
|
|
27
|
+
from dataflow.label.models import DatasetAnnotations, ImageAnnotation, ObjectAnnotation
|
|
29
28
|
|
|
30
29
|
|
|
31
30
|
class FilterAnalyser(BaseAnalyser):
|
|
@@ -99,17 +98,14 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
99
98
|
for new_id, name in new_classes.items():
|
|
100
99
|
if name in name_to_old_id:
|
|
101
100
|
old_id = name_to_old_id[name]
|
|
102
|
-
mapping = CategoryMapping(
|
|
103
|
-
new_id=new_id, old_id=old_id, name=name
|
|
104
|
-
)
|
|
101
|
+
mapping = CategoryMapping(new_id=new_id, old_id=old_id, name=name)
|
|
105
102
|
old_to_new[old_id] = mapping
|
|
106
103
|
kept.append(mapping)
|
|
107
104
|
else:
|
|
108
105
|
missing.append(name)
|
|
109
106
|
if logger:
|
|
110
107
|
logger.warning(
|
|
111
|
-
f'Category "{name}" in new class file not '
|
|
112
|
-
f"found in source — skipping"
|
|
108
|
+
f'Category "{name}" in new class file not found in source — skipping'
|
|
113
109
|
)
|
|
114
110
|
|
|
115
111
|
# Build removed list
|
|
@@ -134,10 +130,7 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
134
130
|
total_after = 0
|
|
135
131
|
|
|
136
132
|
for image_ann in dataset.images:
|
|
137
|
-
filtered = [
|
|
138
|
-
obj for obj in image_ann.objects
|
|
139
|
-
if obj.class_id in old_to_new
|
|
140
|
-
]
|
|
133
|
+
filtered = [obj for obj in image_ann.objects if obj.class_id in old_to_new]
|
|
141
134
|
for obj in filtered:
|
|
142
135
|
mapping = old_to_new[obj.class_id]
|
|
143
136
|
obj.class_id = mapping.new_id
|
|
@@ -202,9 +195,7 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
202
195
|
return result
|
|
203
196
|
|
|
204
197
|
if not new_classes:
|
|
205
|
-
result.add_error(
|
|
206
|
-
f"No valid class names in new class file: {new_class_file}"
|
|
207
|
-
)
|
|
198
|
+
result.add_error(f"No valid class names in new class file: {new_class_file}")
|
|
208
199
|
return result
|
|
209
200
|
|
|
210
201
|
# ---- 3. Detect format + create handler ------------------------
|
|
@@ -259,9 +250,7 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
259
250
|
)
|
|
260
251
|
|
|
261
252
|
if not old_to_new:
|
|
262
|
-
result.add_error(
|
|
263
|
-
"No matching categories between source and new class file"
|
|
264
|
-
)
|
|
253
|
+
result.add_error("No matching categories between source and new class file")
|
|
265
254
|
return result
|
|
266
255
|
|
|
267
256
|
# ---- 5. Ensure output directory --------------------------------
|
|
@@ -304,14 +293,16 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
304
293
|
# rather than mutating the original (avoid
|
|
305
294
|
# aliasing — the original may be reused by the
|
|
306
295
|
# iterator or shared across images).
|
|
307
|
-
filtered_objects.append(
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
296
|
+
filtered_objects.append(
|
|
297
|
+
ObjectAnnotation(
|
|
298
|
+
class_id=mapping.new_id,
|
|
299
|
+
class_name=mapping.name,
|
|
300
|
+
bbox=obj.bbox,
|
|
301
|
+
segmentation=obj.segmentation,
|
|
302
|
+
confidence=obj.confidence,
|
|
303
|
+
is_crowd=obj.is_crowd,
|
|
304
|
+
)
|
|
305
|
+
)
|
|
315
306
|
total_after += len(filtered_objects)
|
|
316
307
|
|
|
317
308
|
if filtered_objects:
|
|
@@ -328,9 +319,7 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
328
319
|
wr = write_handler.write_one(filtered_img, output_dir)
|
|
329
320
|
if not wr.success:
|
|
330
321
|
for err in wr.errors:
|
|
331
|
-
result.add_error(
|
|
332
|
-
f"Write {image_ann.image_id}: {err}"
|
|
333
|
-
)
|
|
322
|
+
result.add_error(f"Write {image_ann.image_id}: {err}")
|
|
334
323
|
return result
|
|
335
324
|
except Exception as e:
|
|
336
325
|
result.add_error(f"Failed during streaming filter: {e}")
|
|
@@ -342,23 +331,19 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
342
331
|
total_files = dataset.num_images
|
|
343
332
|
total_before = dataset.num_objects
|
|
344
333
|
|
|
345
|
-
total_files_with_annotations, total_after = (
|
|
346
|
-
|
|
334
|
+
total_files_with_annotations, total_after = self._filter_dataset_images(
|
|
335
|
+
dataset, old_to_new
|
|
347
336
|
)
|
|
348
337
|
|
|
349
338
|
# Update categories
|
|
350
|
-
dataset.categories = {
|
|
351
|
-
km.new_id: km.name for km in kept_categories
|
|
352
|
-
}
|
|
339
|
+
dataset.categories = {km.new_id: km.name for km in kept_categories}
|
|
353
340
|
|
|
354
341
|
try:
|
|
355
342
|
for image_ann in dataset.images:
|
|
356
343
|
wr = handler.write_one(image_ann, output_dir)
|
|
357
344
|
if not wr.success:
|
|
358
345
|
for err in wr.errors:
|
|
359
|
-
result.add_error(
|
|
360
|
-
f"Write {image_ann.image_id}: {err}"
|
|
361
|
-
)
|
|
346
|
+
result.add_error(f"Write {image_ann.image_id}: {err}")
|
|
362
347
|
return result
|
|
363
348
|
except Exception as e:
|
|
364
349
|
result.add_error(f"Failed to write filtered output: {e}")
|
|
@@ -369,14 +354,12 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
369
354
|
total_files = dataset.num_images
|
|
370
355
|
total_before = dataset.num_objects
|
|
371
356
|
|
|
372
|
-
total_files_with_annotations, total_after = (
|
|
373
|
-
|
|
357
|
+
total_files_with_annotations, total_after = self._filter_dataset_images(
|
|
358
|
+
dataset, old_to_new
|
|
374
359
|
)
|
|
375
360
|
|
|
376
361
|
# Update categories to match new class file
|
|
377
|
-
dataset.categories = {
|
|
378
|
-
km.new_id: km.name for km in kept_categories
|
|
379
|
-
}
|
|
362
|
+
dataset.categories = {km.new_id: km.name for km in kept_categories}
|
|
380
363
|
|
|
381
364
|
output_path = output_dir / label_path.name
|
|
382
365
|
try:
|
|
@@ -413,29 +396,20 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
413
396
|
if missing_categories:
|
|
414
397
|
s = "y" if len(missing_categories) == 1 else "ies"
|
|
415
398
|
result.add_warning(
|
|
416
|
-
f"{len(missing_categories)} categor{s} in new class "
|
|
417
|
-
f"file not found in source"
|
|
399
|
+
f"{len(missing_categories)} categor{s} in new class file not found in source"
|
|
418
400
|
)
|
|
419
401
|
|
|
420
402
|
if total_after == 0 and total_before > 0:
|
|
421
|
-
result.add_warning(
|
|
422
|
-
"All annotations were filtered out — output files are empty"
|
|
423
|
-
)
|
|
403
|
+
result.add_warning("All annotations were filtered out — output files are empty")
|
|
424
404
|
|
|
425
405
|
# ---- 9. Log output ---------------------------------------------
|
|
426
406
|
self._log_info(
|
|
427
|
-
format_analyse_header(
|
|
428
|
-
"Category Filter", label_path, f"{fmt} (auto-detected)"
|
|
429
|
-
)
|
|
407
|
+
format_analyse_header("Category Filter", label_path, f"{fmt} (auto-detected)")
|
|
430
408
|
)
|
|
431
409
|
self._log_info(
|
|
432
|
-
f" Original class: {original_class_file.name} "
|
|
433
|
-
f"({len(original_classes)} categories)"
|
|
434
|
-
)
|
|
435
|
-
self._log_info(
|
|
436
|
-
f" New class: {new_class_file.name} "
|
|
437
|
-
f"({len(new_classes)} categories)\n"
|
|
410
|
+
f" Original class: {original_class_file.name} ({len(original_classes)} categories)"
|
|
438
411
|
)
|
|
412
|
+
self._log_info(f" New class: {new_class_file.name} ({len(new_classes)} categories)\n")
|
|
439
413
|
self._log_info(
|
|
440
414
|
format_filter_result(
|
|
441
415
|
total_files=total_files,
|
|
@@ -449,8 +423,6 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
449
423
|
)
|
|
450
424
|
)
|
|
451
425
|
if result.log_path:
|
|
452
|
-
self._log_info(
|
|
453
|
-
format_analyse_result("✓ Success", result.log_path)
|
|
454
|
-
)
|
|
426
|
+
self._log_info(format_analyse_result("✓ Success", result.log_path))
|
|
455
427
|
|
|
456
428
|
return result
|
|
@@ -55,8 +55,9 @@ def format_analyse_header(
|
|
|
55
55
|
lines.append(format_kv("Source", label))
|
|
56
56
|
else:
|
|
57
57
|
lines.append(
|
|
58
|
-
format_kv(
|
|
59
|
-
|
|
58
|
+
format_kv(
|
|
59
|
+
"Sources", f"{', '.join(str(p) for p in path_list)} ({len(path_list)} paths)"
|
|
60
|
+
)
|
|
60
61
|
)
|
|
61
62
|
|
|
62
63
|
if class_file is not None:
|
|
@@ -81,11 +82,7 @@ def format_stats_path_breakdown(path_stats) -> str:
|
|
|
81
82
|
label = f"{ps['path']}"
|
|
82
83
|
if ps.get("recursive"):
|
|
83
84
|
label += " (recursive)"
|
|
84
|
-
lines.append(
|
|
85
|
-
f" {label:<40} "
|
|
86
|
-
f"{ps['files']:>5} files, "
|
|
87
|
-
f"{ps['annotations']:>5} annotations"
|
|
88
|
-
)
|
|
85
|
+
lines.append(f" {label:<40} {ps['files']:>5} files, {ps['annotations']:>5} annotations")
|
|
89
86
|
lines.append("")
|
|
90
87
|
return "\n".join(lines)
|
|
91
88
|
|
|
@@ -131,19 +128,23 @@ def format_stats_result(
|
|
|
131
128
|
for name, count in per_class.items()
|
|
132
129
|
]
|
|
133
130
|
rows.append(["─" * 15, "─" * 4, "─" * 7])
|
|
134
|
-
rows.append(
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
131
|
+
rows.append(
|
|
132
|
+
[
|
|
133
|
+
f"Total ({len(per_class)})",
|
|
134
|
+
"",
|
|
135
|
+
str(sum(per_class.values())),
|
|
136
|
+
]
|
|
137
|
+
)
|
|
139
138
|
else:
|
|
140
139
|
headers = ["Class", "Count"]
|
|
141
140
|
rows = [[name, str(count)] for name, count in per_class.items()]
|
|
142
141
|
rows.append(["─" * 15, "─" * 7])
|
|
143
|
-
rows.append(
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
142
|
+
rows.append(
|
|
143
|
+
[
|
|
144
|
+
f"Total ({len(per_class)})",
|
|
145
|
+
str(sum(per_class.values())),
|
|
146
|
+
]
|
|
147
|
+
)
|
|
147
148
|
lines.append(format_section("Per-Class"))
|
|
148
149
|
lines.append(format_table(headers, rows))
|
|
149
150
|
else:
|
|
@@ -253,7 +254,7 @@ def format_filter_result(
|
|
|
253
254
|
for name in missing_categories:
|
|
254
255
|
lines.append(f' "{name}"')
|
|
255
256
|
else:
|
|
256
|
-
lines.append(
|
|
257
|
+
lines.append(" Not found in source: 0 categories")
|
|
257
258
|
lines.append("")
|
|
258
259
|
|
|
259
260
|
# ── Filter Summary ──
|
|
@@ -262,10 +263,12 @@ def format_filter_result(
|
|
|
262
263
|
lines.append(format_kv("Files with annotations", str(total_files_with_annotations)))
|
|
263
264
|
lines.append(format_kv("Annotations before", str(annotations_before)))
|
|
264
265
|
lines.append(format_kv("Annotations after", str(annotations_after)))
|
|
265
|
-
lines.append(
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
266
|
+
lines.append(
|
|
267
|
+
format_kv(
|
|
268
|
+
"Output",
|
|
269
|
+
f"{total_files_with_annotations} files → {output_dir}",
|
|
270
|
+
)
|
|
271
|
+
)
|
|
269
272
|
|
|
270
273
|
return "\n".join(lines)
|
|
271
274
|
|
|
@@ -295,14 +298,12 @@ def format_partition_result(
|
|
|
295
298
|
Returns:
|
|
296
299
|
Formatted partition summary.
|
|
297
300
|
"""
|
|
298
|
-
mode_label = {"images": "Images only", "labels": "Labels only",
|
|
299
|
-
"both": "Labels + Images"}[mode]
|
|
301
|
+
mode_label = {"images": "Images only", "labels": "Labels only", "both": "Labels + Images"}[mode]
|
|
300
302
|
|
|
301
303
|
lines = [
|
|
302
304
|
format_kv("Mode", mode_label),
|
|
303
305
|
format_kv("Partitions", str(num_partitions)),
|
|
304
|
-
format_kv("Shuffle", f"{'Yes' if shuffle else 'No'}"
|
|
305
|
-
f"{f' (seed={seed})' if shuffle else ''}"),
|
|
306
|
+
format_kv("Shuffle", f"{'Yes' if shuffle else 'No'}{f' (seed={seed})' if shuffle else ''}"),
|
|
306
307
|
format_kv("Move", "Yes" if move else "No"),
|
|
307
308
|
format_kv("Total files", str(total_files)),
|
|
308
309
|
"",
|
|
@@ -310,10 +311,7 @@ def format_partition_result(
|
|
|
310
311
|
]
|
|
311
312
|
|
|
312
313
|
for i in range(num_partitions):
|
|
313
|
-
lines.append(
|
|
314
|
-
f" Part {i + 1}: {partition_sizes[i]:>6} files → "
|
|
315
|
-
f"{partition_dirs[i]}"
|
|
316
|
-
)
|
|
314
|
+
lines.append(f" Part {i + 1}: {partition_sizes[i]:>6} files → {partition_dirs[i]}")
|
|
317
315
|
|
|
318
316
|
return "\n".join(lines)
|
|
319
317
|
|
|
@@ -26,7 +26,6 @@ from .utils import (
|
|
|
26
26
|
_IMAGE_EXTENSIONS,
|
|
27
27
|
create_handler,
|
|
28
28
|
detect_format,
|
|
29
|
-
load_class_names,
|
|
30
29
|
)
|
|
31
30
|
|
|
32
31
|
|
|
@@ -93,15 +92,11 @@ class PartitionAnalyser(BaseAnalyser):
|
|
|
93
92
|
# 1. Validate inputs
|
|
94
93
|
# ------------------------------------------------------------------
|
|
95
94
|
if num < 2:
|
|
96
|
-
result.add_error(
|
|
97
|
-
f"Number of partitions must be at least 2, got: {num}"
|
|
98
|
-
)
|
|
95
|
+
result.add_error(f"Number of partitions must be at least 2, got: {num}")
|
|
99
96
|
return result
|
|
100
97
|
|
|
101
98
|
if label_dir is None and image_dir is None:
|
|
102
|
-
result.add_error(
|
|
103
|
-
"At least one of label_dir or image_dir must be provided"
|
|
104
|
-
)
|
|
99
|
+
result.add_error("At least one of label_dir or image_dir must be provided")
|
|
105
100
|
return result
|
|
106
101
|
|
|
107
102
|
# ------------------------------------------------------------------
|
|
@@ -246,15 +241,9 @@ class PartitionAnalyser(BaseAnalyser):
|
|
|
246
241
|
wr = handler.write_one(img_ann, part_dir)
|
|
247
242
|
if not wr.success:
|
|
248
243
|
for err in wr.errors:
|
|
249
|
-
result.add_error(
|
|
250
|
-
f"Write {part_dir.name}/"
|
|
251
|
-
f"{img_ann.image_id}: {err}"
|
|
252
|
-
)
|
|
244
|
+
result.add_error(f"Write {part_dir.name}/{img_ann.image_id}: {err}")
|
|
253
245
|
except Exception as e:
|
|
254
|
-
result.add_error(
|
|
255
|
-
f"Write {part_dir.name}/"
|
|
256
|
-
f"{img_ann.image_id}: {e}"
|
|
257
|
-
)
|
|
246
|
+
result.add_error(f"Write {part_dir.name}/{img_ann.image_id}: {e}")
|
|
258
247
|
|
|
259
248
|
# Move label files if in move mode
|
|
260
249
|
if move:
|
|
@@ -265,9 +254,7 @@ class PartitionAnalyser(BaseAnalyser):
|
|
|
265
254
|
else: # labelme
|
|
266
255
|
src_label = label_dir / f"{img_ann.image_id}.json"
|
|
267
256
|
if src_label.exists():
|
|
268
|
-
_copy_or_move_file(
|
|
269
|
-
src_label, part_dir, move=True, logger=self.logger
|
|
270
|
-
)
|
|
257
|
+
_copy_or_move_file(src_label, part_dir, move=True, logger=self.logger)
|
|
271
258
|
|
|
272
259
|
elif mode == "both":
|
|
273
260
|
labels_subdir = part_dir / "labels"
|
|
@@ -282,25 +269,18 @@ class PartitionAnalyser(BaseAnalyser):
|
|
|
282
269
|
if not wr.success:
|
|
283
270
|
for err in wr.errors:
|
|
284
271
|
result.add_error(
|
|
285
|
-
f"Write {part_dir.name}/labels/"
|
|
286
|
-
f"{img_ann.image_id}: {err}"
|
|
272
|
+
f"Write {part_dir.name}/labels/{img_ann.image_id}: {err}"
|
|
287
273
|
)
|
|
288
274
|
except Exception as e:
|
|
289
|
-
result.add_error(
|
|
290
|
-
f"Write {part_dir.name}/labels/"
|
|
291
|
-
f"{img_ann.image_id}: {e}"
|
|
292
|
-
)
|
|
275
|
+
result.add_error(f"Write {part_dir.name}/labels/{img_ann.image_id}: {e}")
|
|
293
276
|
|
|
294
277
|
# Match and copy/move image
|
|
295
278
|
stem = img_ann.image_id
|
|
296
279
|
if stem in image_stems:
|
|
297
|
-
_copy_or_move_file(
|
|
298
|
-
image_stems[stem], images_subdir, move, self.logger
|
|
299
|
-
)
|
|
280
|
+
_copy_or_move_file(image_stems[stem], images_subdir, move, self.logger)
|
|
300
281
|
else:
|
|
301
282
|
self._log_warning(
|
|
302
|
-
f"No matching image found for label "
|
|
303
|
-
f"'{img_ann.image_id}' in {image_dir}"
|
|
283
|
+
f"No matching image found for label '{img_ann.image_id}' in {image_dir}"
|
|
304
284
|
)
|
|
305
285
|
|
|
306
286
|
# Move label files if in move mode
|
|
@@ -312,21 +292,16 @@ class PartitionAnalyser(BaseAnalyser):
|
|
|
312
292
|
src_label = label_dir / f"{img_ann.image_id}.json"
|
|
313
293
|
if src_label.exists():
|
|
314
294
|
_copy_or_move_file(
|
|
315
|
-
src_label, labels_subdir,
|
|
316
|
-
move=True, logger=self.logger
|
|
295
|
+
src_label, labels_subdir, move=True, logger=self.logger
|
|
317
296
|
)
|
|
318
297
|
|
|
319
298
|
# Report unmatched images (in image_dir but not in labels)
|
|
320
299
|
if not move: # Only warn for copy mode; move mode self-resolves
|
|
321
|
-
label_stems = {
|
|
322
|
-
img_ann.image_id
|
|
323
|
-
for img_ann in items[start:end]
|
|
324
|
-
}
|
|
300
|
+
label_stems = {img_ann.image_id for img_ann in items[start:end]}
|
|
325
301
|
for stem, img_path in image_stems.items():
|
|
326
302
|
if stem not in label_stems:
|
|
327
303
|
self._log_warning(
|
|
328
|
-
f"Image '{img_path.name}' has no matching "
|
|
329
|
-
f"label — skipped"
|
|
304
|
+
f"Image '{img_path.name}' has no matching label — skipped"
|
|
330
305
|
)
|
|
331
306
|
|
|
332
307
|
# Copy class_file to partition directory
|
|
@@ -336,10 +311,7 @@ class PartitionAnalyser(BaseAnalyser):
|
|
|
336
311
|
if not target_cf.exists():
|
|
337
312
|
shutil.copy2(str(class_file), str(target_cf))
|
|
338
313
|
except OSError as e:
|
|
339
|
-
result.add_warning(
|
|
340
|
-
f"Could not copy class file to "
|
|
341
|
-
f"{part_dir.name}: {e}"
|
|
342
|
-
)
|
|
314
|
+
result.add_warning(f"Could not copy class file to {part_dir.name}: {e}")
|
|
343
315
|
|
|
344
316
|
# ------------------------------------------------------------------
|
|
345
317
|
# 6. Build result
|
|
@@ -361,8 +333,9 @@ class PartitionAnalyser(BaseAnalyser):
|
|
|
361
333
|
# ------------------------------------------------------------------
|
|
362
334
|
# 7. Log output
|
|
363
335
|
# ------------------------------------------------------------------
|
|
364
|
-
mode_label = {"images": "Images Only", "labels": "Labels Only",
|
|
365
|
-
|
|
336
|
+
mode_label = {"images": "Images Only", "labels": "Labels Only", "both": "Labels + Images"}[
|
|
337
|
+
mode
|
|
338
|
+
]
|
|
366
339
|
label_paths = label_dir if label_dir else image_dir
|
|
367
340
|
self._log_info(
|
|
368
341
|
format_analyse_header(
|
|
@@ -385,8 +358,6 @@ class PartitionAnalyser(BaseAnalyser):
|
|
|
385
358
|
)
|
|
386
359
|
)
|
|
387
360
|
if result.log_path:
|
|
388
|
-
self._log_info(
|
|
389
|
-
format_analyse_result("✓ Success", result.log_path)
|
|
390
|
-
)
|
|
361
|
+
self._log_info(format_analyse_result("✓ Success", result.log_path))
|
|
391
362
|
|
|
392
363
|
return result
|
|
@@ -103,15 +103,11 @@ class SampleAnalyser(BaseAnalyser):
|
|
|
103
103
|
# 1. Validate inputs
|
|
104
104
|
# ------------------------------------------------------------------
|
|
105
105
|
if label_dir is None and image_dir is None:
|
|
106
|
-
result.add_error(
|
|
107
|
-
"At least one of label_dir or image_dir must be provided"
|
|
108
|
-
)
|
|
106
|
+
result.add_error("At least one of label_dir or image_dir must be provided")
|
|
109
107
|
return result
|
|
110
108
|
|
|
111
109
|
if count < 1:
|
|
112
|
-
result.add_error(
|
|
113
|
-
f"Count must be at least 1, got: {count}"
|
|
114
|
-
)
|
|
110
|
+
result.add_error(f"Count must be at least 1, got: {count}")
|
|
115
111
|
return result
|
|
116
112
|
|
|
117
113
|
# ------------------------------------------------------------------
|
|
@@ -190,8 +186,7 @@ class SampleAnalyser(BaseAnalyser):
|
|
|
190
186
|
actual_count = min(count, total)
|
|
191
187
|
if count > total:
|
|
192
188
|
result.add_warning(
|
|
193
|
-
f"Requested {count} files but only {total} available — "
|
|
194
|
-
f"collecting all"
|
|
189
|
+
f"Requested {count} files but only {total} available — collecting all"
|
|
195
190
|
)
|
|
196
191
|
|
|
197
192
|
sampled = items[:actual_count]
|
|
@@ -221,13 +216,10 @@ class SampleAnalyser(BaseAnalyser):
|
|
|
221
216
|
_copy_or_move_file(src_path, label_subdir, move, self.logger)
|
|
222
217
|
# Match and copy image
|
|
223
218
|
if stem in image_stems:
|
|
224
|
-
_copy_or_move_file(
|
|
225
|
-
image_stems[stem], image_subdir, move, self.logger
|
|
226
|
-
)
|
|
219
|
+
_copy_or_move_file(image_stems[stem], image_subdir, move, self.logger)
|
|
227
220
|
else:
|
|
228
221
|
self._log_warning(
|
|
229
|
-
f"No matching image found for label '{stem}' "
|
|
230
|
-
f"in image directory"
|
|
222
|
+
f"No matching image found for label '{stem}' in image directory"
|
|
231
223
|
)
|
|
232
224
|
unmatched_image_warnings += 1
|
|
233
225
|
else:
|
|
@@ -238,10 +230,7 @@ class SampleAnalyser(BaseAnalyser):
|
|
|
238
230
|
sampled_stems = {stem for _, stem in sampled}
|
|
239
231
|
for stem, img_path in image_stems.items():
|
|
240
232
|
if stem not in sampled_stems:
|
|
241
|
-
self._log_warning(
|
|
242
|
-
f"Image '{img_path.name}' has no matching "
|
|
243
|
-
f"label — skipped"
|
|
244
|
-
)
|
|
233
|
+
self._log_warning(f"Image '{img_path.name}' has no matching label — skipped")
|
|
245
234
|
|
|
246
235
|
# ------------------------------------------------------------------
|
|
247
236
|
# 7. Copy class_file to output_dir if provided
|
|
@@ -255,9 +244,7 @@ class SampleAnalyser(BaseAnalyser):
|
|
|
255
244
|
else:
|
|
256
245
|
shutil.copy2(str(class_file), str(target_cf))
|
|
257
246
|
except OSError as e:
|
|
258
|
-
result.add_warning(
|
|
259
|
-
f"Could not copy class file: {e}"
|
|
260
|
-
)
|
|
247
|
+
result.add_warning(f"Could not copy class file: {e}")
|
|
261
248
|
|
|
262
249
|
# ------------------------------------------------------------------
|
|
263
250
|
# 8. Build result
|
|
@@ -306,8 +293,6 @@ class SampleAnalyser(BaseAnalyser):
|
|
|
306
293
|
)
|
|
307
294
|
)
|
|
308
295
|
if result.log_path:
|
|
309
|
-
self._log_info(
|
|
310
|
-
format_analyse_result("✓ Success", result.log_path)
|
|
311
|
-
)
|
|
296
|
+
self._log_info(format_analyse_result("✓ Success", result.log_path))
|
|
312
297
|
|
|
313
298
|
return result
|
|
@@ -101,15 +101,11 @@ class SplitAnalyser(BaseAnalyser):
|
|
|
101
101
|
# 1. Validate inputs
|
|
102
102
|
# ------------------------------------------------------------------
|
|
103
103
|
if label_dir is None and image_dir is None:
|
|
104
|
-
result.add_error(
|
|
105
|
-
"At least one of label_dir or image_dir must be provided"
|
|
106
|
-
)
|
|
104
|
+
result.add_error("At least one of label_dir or image_dir must be provided")
|
|
107
105
|
return result
|
|
108
106
|
|
|
109
107
|
if not 0.0 < ratio < 1.0:
|
|
110
|
-
result.add_error(
|
|
111
|
-
f"Ratio must be between 0 and 1 (exclusive), got: {ratio}"
|
|
112
|
-
)
|
|
108
|
+
result.add_error(f"Ratio must be between 0 and 1 (exclusive), got: {ratio}")
|
|
113
109
|
return result
|
|
114
110
|
|
|
115
111
|
# ------------------------------------------------------------------
|
|
@@ -238,10 +234,7 @@ class SplitAnalyser(BaseAnalyser):
|
|
|
238
234
|
label_stems = {stem for _, stem in train_items + val_items}
|
|
239
235
|
for stem, img_path in image_stems.items():
|
|
240
236
|
if stem not in label_stems:
|
|
241
|
-
self._log_warning(
|
|
242
|
-
f"Image '{img_path.name}' has no matching "
|
|
243
|
-
f"label — skipped"
|
|
244
|
-
)
|
|
237
|
+
self._log_warning(f"Image '{img_path.name}' has no matching label — skipped")
|
|
245
238
|
|
|
246
239
|
# ------------------------------------------------------------------
|
|
247
240
|
# 7. Copy class_file to both output dirs if provided
|
|
@@ -302,9 +295,7 @@ class SplitAnalyser(BaseAnalyser):
|
|
|
302
295
|
)
|
|
303
296
|
)
|
|
304
297
|
if result.log_path:
|
|
305
|
-
self._log_info(
|
|
306
|
-
format_analyse_result("✓ Success", result.log_path)
|
|
307
|
-
)
|
|
298
|
+
self._log_info(format_analyse_result("✓ Success", result.log_path))
|
|
308
299
|
|
|
309
300
|
return result
|
|
310
301
|
|
|
@@ -343,26 +334,16 @@ def _split_files(
|
|
|
343
334
|
for label_path, stem in train_items:
|
|
344
335
|
_copy_or_move_file(label_path, train_label_dir, move, logger)
|
|
345
336
|
if stem in image_stems:
|
|
346
|
-
_copy_or_move_file(
|
|
347
|
-
image_stems[stem], train_image_dir, move, logger
|
|
348
|
-
)
|
|
337
|
+
_copy_or_move_file(image_stems[stem], train_image_dir, move, logger)
|
|
349
338
|
else:
|
|
350
|
-
logger.warning(
|
|
351
|
-
f"No matching image found for label '{stem}' in "
|
|
352
|
-
f"image directory"
|
|
353
|
-
)
|
|
339
|
+
logger.warning(f"No matching image found for label '{stem}' in image directory")
|
|
354
340
|
|
|
355
341
|
for label_path, stem in val_items:
|
|
356
342
|
_copy_or_move_file(label_path, val_label_dir, move, logger)
|
|
357
343
|
if stem in image_stems:
|
|
358
|
-
_copy_or_move_file(
|
|
359
|
-
image_stems[stem], val_image_dir, move, logger
|
|
360
|
-
)
|
|
344
|
+
_copy_or_move_file(image_stems[stem], val_image_dir, move, logger)
|
|
361
345
|
else:
|
|
362
|
-
logger.warning(
|
|
363
|
-
f"No matching image found for label '{stem}' in "
|
|
364
|
-
f"image directory"
|
|
365
|
-
)
|
|
346
|
+
logger.warning(f"No matching image found for label '{stem}' in image directory")
|
|
366
347
|
else:
|
|
367
348
|
train_dir = output_dir / "train"
|
|
368
349
|
val_dir = output_dir / "val"
|