dataflow-cv 2.0.0__tar.gz → 2.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/PKG-INFO +3 -4
  2. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/README.md +1 -0
  3. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/__init__.py +1 -1
  4. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/filter.py +30 -58
  5. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/log_templates.py +27 -29
  6. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/partition.py +17 -46
  7. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/sample.py +8 -23
  8. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/split.py +8 -27
  9. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/stats.py +23 -21
  10. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/utils.py +12 -31
  11. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/__init__.py +1 -1
  12. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/__init__.py +1 -1
  13. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/analyse.py +18 -34
  14. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/convert.py +28 -15
  15. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/evaluate.py +16 -4
  16. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/utils.py +12 -10
  17. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/commands/visualize.py +15 -7
  18. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/exceptions.py +1 -1
  19. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/cli/main.py +14 -9
  20. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/base.py +49 -74
  21. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/coco_and_labelme.py +31 -30
  22. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/labelme_and_yolo.py +30 -40
  23. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/rle_converter.py +3 -10
  24. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/utils.py +16 -42
  25. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/yolo_and_coco.py +33 -44
  26. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/base.py +6 -16
  27. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/evaluator.py +1 -1
  28. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/log_templates.py +40 -14
  29. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/metrics.py +28 -32
  30. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/result.py +23 -26
  31. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/utils.py +13 -32
  32. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/__init__.py +8 -2
  33. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/base.py +20 -48
  34. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/coco_handler.py +48 -82
  35. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/labelme_handler.py +51 -99
  36. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/models.py +7 -21
  37. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/utils.py +2 -6
  38. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/label/yolo_handler.py +78 -143
  39. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/util/logging.py +8 -28
  40. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/__init__.py +1 -2
  41. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/base.py +21 -60
  42. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/coco_visualizer.py +4 -10
  43. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/labelme_visualizer.py +2 -7
  44. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/log_templates.py +1 -1
  45. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/utils.py +0 -1
  46. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/visualize/yolo_visualizer.py +4 -12
  47. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/PKG-INFO +3 -4
  48. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/requires.txt +1 -3
  49. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/pyproject.toml +10 -26
  50. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/LICENSE +0 -0
  51. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/__init__.py +0 -0
  52. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/analyse/base.py +0 -0
  53. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/__init__.py +0 -0
  54. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/convert/log_templates.py +0 -0
  55. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/evaluate/__init__.py +0 -0
  56. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow/util/__init__.py +0 -0
  57. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/SOURCES.txt +0 -0
  58. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/dependency_links.txt +0 -0
  59. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/entry_points.txt +0 -0
  60. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/not-zip-safe +0 -0
  61. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/dataflow_cv.egg-info/top_level.txt +0 -0
  62. {dataflow_cv-2.0.0 → dataflow_cv-2.0.1}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dataflow-cv
3
- Version: 2.0.0
3
+ Version: 2.0.1
4
4
  Summary: A computer vision dataset processing library — analyse, convert, visualize, and evaluate annotations across YOLO, LabelMe, and COCO formats
5
5
  Author: DataFlow-CV Team
6
6
  License: MIT
@@ -33,9 +33,7 @@ Requires-Dist: pycocotools>=2.0.0; extra == "coco"
33
33
  Provides-Extra: dev
34
34
  Requires-Dist: pytest>=7.0.0; extra == "dev"
35
35
  Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
36
- Requires-Dist: black>=22.0.0; extra == "dev"
37
- Requires-Dist: isort>=5.12.0; extra == "dev"
38
- Requires-Dist: flake8>=6.0.0; extra == "dev"
36
+ Requires-Dist: ruff>=0.16; extra == "dev"
39
37
  Requires-Dist: mypy>=1.0.0; extra == "dev"
40
38
  Dynamic: license-file
41
39
 
@@ -67,6 +65,7 @@ A computer vision dataset processing library — analyse, convert, visualize, an
67
65
  | 🎨 **Visualize** | OpenCV rendering with color-coded classes, display & save modes | `dataflow-cv visualize yolo ...` |
68
66
  | 📊 **Evaluate** | COCO mAP via pycocotools, single-threshold P/R/F1 per class | `dataflow-cv evaluate detection ...` |
69
67
  | 💻 **CLI + API** | Click-based CLI with rich `--help`; Python API for pipelines | `from dataflow.convert import ...` |
68
+ | 🤖 **AI Skills** | Claude Code skill (`/dataflow:dataflow-cv`) for AI assistants — CLI/API reference & known gotchas | `claude plugin install dataflow@claude-skills` |
70
69
 
71
70
  ---
72
71
 
@@ -26,6 +26,7 @@ A computer vision dataset processing library — analyse, convert, visualize, an
26
26
  | 🎨 **Visualize** | OpenCV rendering with color-coded classes, display & save modes | `dataflow-cv visualize yolo ...` |
27
27
  | 📊 **Evaluate** | COCO mAP via pycocotools, single-threshold P/R/F1 per class | `dataflow-cv evaluate detection ...` |
28
28
  | 💻 **CLI + API** | Click-based CLI with rich `--help`; Python API for pipelines | `from dataflow.convert import ...` |
29
+ | 🤖 **AI Skills** | Claude Code skill (`/dataflow:dataflow-cv`) for AI assistants — CLI/API reference & known gotchas | `claude plugin install dataflow@claude-skills` |
29
30
 
30
31
  ---
31
32
 
@@ -1,4 +1,4 @@
1
1
  """DataFlow-CV: A computer vision dataset processing library."""
2
2
 
3
- __version__ = "2.0.0"
3
+ __version__ = "2.0.1"
4
4
  __author__ = "DataFlow-CV Team"
@@ -24,8 +24,7 @@ from .log_templates import (
24
24
  format_filter_result,
25
25
  )
26
26
  from .utils import create_handler, detect_format, load_class_names
27
- from dataflow.label.models import (DatasetAnnotations, ImageAnnotation,
28
- ObjectAnnotation)
27
+ from dataflow.label.models import DatasetAnnotations, ImageAnnotation, ObjectAnnotation
29
28
 
30
29
 
31
30
  class FilterAnalyser(BaseAnalyser):
@@ -99,17 +98,14 @@ class FilterAnalyser(BaseAnalyser):
99
98
  for new_id, name in new_classes.items():
100
99
  if name in name_to_old_id:
101
100
  old_id = name_to_old_id[name]
102
- mapping = CategoryMapping(
103
- new_id=new_id, old_id=old_id, name=name
104
- )
101
+ mapping = CategoryMapping(new_id=new_id, old_id=old_id, name=name)
105
102
  old_to_new[old_id] = mapping
106
103
  kept.append(mapping)
107
104
  else:
108
105
  missing.append(name)
109
106
  if logger:
110
107
  logger.warning(
111
- f'Category "{name}" in new class file not '
112
- f"found in source — skipping"
108
+ f'Category "{name}" in new class file not found in source — skipping'
113
109
  )
114
110
 
115
111
  # Build removed list
@@ -134,10 +130,7 @@ class FilterAnalyser(BaseAnalyser):
134
130
  total_after = 0
135
131
 
136
132
  for image_ann in dataset.images:
137
- filtered = [
138
- obj for obj in image_ann.objects
139
- if obj.class_id in old_to_new
140
- ]
133
+ filtered = [obj for obj in image_ann.objects if obj.class_id in old_to_new]
141
134
  for obj in filtered:
142
135
  mapping = old_to_new[obj.class_id]
143
136
  obj.class_id = mapping.new_id
@@ -202,9 +195,7 @@ class FilterAnalyser(BaseAnalyser):
202
195
  return result
203
196
 
204
197
  if not new_classes:
205
- result.add_error(
206
- f"No valid class names in new class file: {new_class_file}"
207
- )
198
+ result.add_error(f"No valid class names in new class file: {new_class_file}")
208
199
  return result
209
200
 
210
201
  # ---- 3. Detect format + create handler ------------------------
@@ -259,9 +250,7 @@ class FilterAnalyser(BaseAnalyser):
259
250
  )
260
251
 
261
252
  if not old_to_new:
262
- result.add_error(
263
- "No matching categories between source and new class file"
264
- )
253
+ result.add_error("No matching categories between source and new class file")
265
254
  return result
266
255
 
267
256
  # ---- 5. Ensure output directory --------------------------------
@@ -304,14 +293,16 @@ class FilterAnalyser(BaseAnalyser):
304
293
  # rather than mutating the original (avoid
305
294
  # aliasing — the original may be reused by the
306
295
  # iterator or shared across images).
307
- filtered_objects.append(ObjectAnnotation(
308
- class_id=mapping.new_id,
309
- class_name=mapping.name,
310
- bbox=obj.bbox,
311
- segmentation=obj.segmentation,
312
- confidence=obj.confidence,
313
- is_crowd=obj.is_crowd,
314
- ))
296
+ filtered_objects.append(
297
+ ObjectAnnotation(
298
+ class_id=mapping.new_id,
299
+ class_name=mapping.name,
300
+ bbox=obj.bbox,
301
+ segmentation=obj.segmentation,
302
+ confidence=obj.confidence,
303
+ is_crowd=obj.is_crowd,
304
+ )
305
+ )
315
306
  total_after += len(filtered_objects)
316
307
 
317
308
  if filtered_objects:
@@ -328,9 +319,7 @@ class FilterAnalyser(BaseAnalyser):
328
319
  wr = write_handler.write_one(filtered_img, output_dir)
329
320
  if not wr.success:
330
321
  for err in wr.errors:
331
- result.add_error(
332
- f"Write {image_ann.image_id}: {err}"
333
- )
322
+ result.add_error(f"Write {image_ann.image_id}: {err}")
334
323
  return result
335
324
  except Exception as e:
336
325
  result.add_error(f"Failed during streaming filter: {e}")
@@ -342,23 +331,19 @@ class FilterAnalyser(BaseAnalyser):
342
331
  total_files = dataset.num_images
343
332
  total_before = dataset.num_objects
344
333
 
345
- total_files_with_annotations, total_after = (
346
- self._filter_dataset_images(dataset, old_to_new)
334
+ total_files_with_annotations, total_after = self._filter_dataset_images(
335
+ dataset, old_to_new
347
336
  )
348
337
 
349
338
  # Update categories
350
- dataset.categories = {
351
- km.new_id: km.name for km in kept_categories
352
- }
339
+ dataset.categories = {km.new_id: km.name for km in kept_categories}
353
340
 
354
341
  try:
355
342
  for image_ann in dataset.images:
356
343
  wr = handler.write_one(image_ann, output_dir)
357
344
  if not wr.success:
358
345
  for err in wr.errors:
359
- result.add_error(
360
- f"Write {image_ann.image_id}: {err}"
361
- )
346
+ result.add_error(f"Write {image_ann.image_id}: {err}")
362
347
  return result
363
348
  except Exception as e:
364
349
  result.add_error(f"Failed to write filtered output: {e}")
@@ -369,14 +354,12 @@ class FilterAnalyser(BaseAnalyser):
369
354
  total_files = dataset.num_images
370
355
  total_before = dataset.num_objects
371
356
 
372
- total_files_with_annotations, total_after = (
373
- self._filter_dataset_images(dataset, old_to_new)
357
+ total_files_with_annotations, total_after = self._filter_dataset_images(
358
+ dataset, old_to_new
374
359
  )
375
360
 
376
361
  # Update categories to match new class file
377
- dataset.categories = {
378
- km.new_id: km.name for km in kept_categories
379
- }
362
+ dataset.categories = {km.new_id: km.name for km in kept_categories}
380
363
 
381
364
  output_path = output_dir / label_path.name
382
365
  try:
@@ -413,29 +396,20 @@ class FilterAnalyser(BaseAnalyser):
413
396
  if missing_categories:
414
397
  s = "y" if len(missing_categories) == 1 else "ies"
415
398
  result.add_warning(
416
- f"{len(missing_categories)} categor{s} in new class "
417
- f"file not found in source"
399
+ f"{len(missing_categories)} categor{s} in new class file not found in source"
418
400
  )
419
401
 
420
402
  if total_after == 0 and total_before > 0:
421
- result.add_warning(
422
- "All annotations were filtered out — output files are empty"
423
- )
403
+ result.add_warning("All annotations were filtered out — output files are empty")
424
404
 
425
405
  # ---- 9. Log output ---------------------------------------------
426
406
  self._log_info(
427
- format_analyse_header(
428
- "Category Filter", label_path, f"{fmt} (auto-detected)"
429
- )
407
+ format_analyse_header("Category Filter", label_path, f"{fmt} (auto-detected)")
430
408
  )
431
409
  self._log_info(
432
- f" Original class: {original_class_file.name} "
433
- f"({len(original_classes)} categories)"
434
- )
435
- self._log_info(
436
- f" New class: {new_class_file.name} "
437
- f"({len(new_classes)} categories)\n"
410
+ f" Original class: {original_class_file.name} ({len(original_classes)} categories)"
438
411
  )
412
+ self._log_info(f" New class: {new_class_file.name} ({len(new_classes)} categories)\n")
439
413
  self._log_info(
440
414
  format_filter_result(
441
415
  total_files=total_files,
@@ -449,8 +423,6 @@ class FilterAnalyser(BaseAnalyser):
449
423
  )
450
424
  )
451
425
  if result.log_path:
452
- self._log_info(
453
- format_analyse_result("✓ Success", result.log_path)
454
- )
426
+ self._log_info(format_analyse_result("✓ Success", result.log_path))
455
427
 
456
428
  return result
@@ -55,8 +55,9 @@ def format_analyse_header(
55
55
  lines.append(format_kv("Source", label))
56
56
  else:
57
57
  lines.append(
58
- format_kv("Sources", f"{', '.join(str(p) for p in path_list)}"
59
- f" ({len(path_list)} paths)")
58
+ format_kv(
59
+ "Sources", f"{', '.join(str(p) for p in path_list)} ({len(path_list)} paths)"
60
+ )
60
61
  )
61
62
 
62
63
  if class_file is not None:
@@ -81,11 +82,7 @@ def format_stats_path_breakdown(path_stats) -> str:
81
82
  label = f"{ps['path']}"
82
83
  if ps.get("recursive"):
83
84
  label += " (recursive)"
84
- lines.append(
85
- f" {label:<40} "
86
- f"{ps['files']:>5} files, "
87
- f"{ps['annotations']:>5} annotations"
88
- )
85
+ lines.append(f" {label:<40} {ps['files']:>5} files, {ps['annotations']:>5} annotations")
89
86
  lines.append("")
90
87
  return "\n".join(lines)
91
88
 
@@ -131,19 +128,23 @@ def format_stats_result(
131
128
  for name, count in per_class.items()
132
129
  ]
133
130
  rows.append(["─" * 15, "─" * 4, "─" * 7])
134
- rows.append([
135
- f"Total ({len(per_class)})",
136
- "",
137
- str(sum(per_class.values())),
138
- ])
131
+ rows.append(
132
+ [
133
+ f"Total ({len(per_class)})",
134
+ "",
135
+ str(sum(per_class.values())),
136
+ ]
137
+ )
139
138
  else:
140
139
  headers = ["Class", "Count"]
141
140
  rows = [[name, str(count)] for name, count in per_class.items()]
142
141
  rows.append(["─" * 15, "─" * 7])
143
- rows.append([
144
- f"Total ({len(per_class)})",
145
- str(sum(per_class.values())),
146
- ])
142
+ rows.append(
143
+ [
144
+ f"Total ({len(per_class)})",
145
+ str(sum(per_class.values())),
146
+ ]
147
+ )
147
148
  lines.append(format_section("Per-Class"))
148
149
  lines.append(format_table(headers, rows))
149
150
  else:
@@ -253,7 +254,7 @@ def format_filter_result(
253
254
  for name in missing_categories:
254
255
  lines.append(f' "{name}"')
255
256
  else:
256
- lines.append(f" Not found in source: 0 categories")
257
+ lines.append(" Not found in source: 0 categories")
257
258
  lines.append("")
258
259
 
259
260
  # ── Filter Summary ──
@@ -262,10 +263,12 @@ def format_filter_result(
262
263
  lines.append(format_kv("Files with annotations", str(total_files_with_annotations)))
263
264
  lines.append(format_kv("Annotations before", str(annotations_before)))
264
265
  lines.append(format_kv("Annotations after", str(annotations_after)))
265
- lines.append(format_kv(
266
- "Output",
267
- f"{total_files_with_annotations} files → {output_dir}",
268
- ))
266
+ lines.append(
267
+ format_kv(
268
+ "Output",
269
+ f"{total_files_with_annotations} files → {output_dir}",
270
+ )
271
+ )
269
272
 
270
273
  return "\n".join(lines)
271
274
 
@@ -295,14 +298,12 @@ def format_partition_result(
295
298
  Returns:
296
299
  Formatted partition summary.
297
300
  """
298
- mode_label = {"images": "Images only", "labels": "Labels only",
299
- "both": "Labels + Images"}[mode]
301
+ mode_label = {"images": "Images only", "labels": "Labels only", "both": "Labels + Images"}[mode]
300
302
 
301
303
  lines = [
302
304
  format_kv("Mode", mode_label),
303
305
  format_kv("Partitions", str(num_partitions)),
304
- format_kv("Shuffle", f"{'Yes' if shuffle else 'No'}"
305
- f"{f' (seed={seed})' if shuffle else ''}"),
306
+ format_kv("Shuffle", f"{'Yes' if shuffle else 'No'}{f' (seed={seed})' if shuffle else ''}"),
306
307
  format_kv("Move", "Yes" if move else "No"),
307
308
  format_kv("Total files", str(total_files)),
308
309
  "",
@@ -310,10 +311,7 @@ def format_partition_result(
310
311
  ]
311
312
 
312
313
  for i in range(num_partitions):
313
- lines.append(
314
- f" Part {i + 1}: {partition_sizes[i]:>6} files → "
315
- f"{partition_dirs[i]}"
316
- )
314
+ lines.append(f" Part {i + 1}: {partition_sizes[i]:>6} files → {partition_dirs[i]}")
317
315
 
318
316
  return "\n".join(lines)
319
317
 
@@ -26,7 +26,6 @@ from .utils import (
26
26
  _IMAGE_EXTENSIONS,
27
27
  create_handler,
28
28
  detect_format,
29
- load_class_names,
30
29
  )
31
30
 
32
31
 
@@ -93,15 +92,11 @@ class PartitionAnalyser(BaseAnalyser):
93
92
  # 1. Validate inputs
94
93
  # ------------------------------------------------------------------
95
94
  if num < 2:
96
- result.add_error(
97
- f"Number of partitions must be at least 2, got: {num}"
98
- )
95
+ result.add_error(f"Number of partitions must be at least 2, got: {num}")
99
96
  return result
100
97
 
101
98
  if label_dir is None and image_dir is None:
102
- result.add_error(
103
- "At least one of label_dir or image_dir must be provided"
104
- )
99
+ result.add_error("At least one of label_dir or image_dir must be provided")
105
100
  return result
106
101
 
107
102
  # ------------------------------------------------------------------
@@ -246,15 +241,9 @@ class PartitionAnalyser(BaseAnalyser):
246
241
  wr = handler.write_one(img_ann, part_dir)
247
242
  if not wr.success:
248
243
  for err in wr.errors:
249
- result.add_error(
250
- f"Write {part_dir.name}/"
251
- f"{img_ann.image_id}: {err}"
252
- )
244
+ result.add_error(f"Write {part_dir.name}/{img_ann.image_id}: {err}")
253
245
  except Exception as e:
254
- result.add_error(
255
- f"Write {part_dir.name}/"
256
- f"{img_ann.image_id}: {e}"
257
- )
246
+ result.add_error(f"Write {part_dir.name}/{img_ann.image_id}: {e}")
258
247
 
259
248
  # Move label files if in move mode
260
249
  if move:
@@ -265,9 +254,7 @@ class PartitionAnalyser(BaseAnalyser):
265
254
  else: # labelme
266
255
  src_label = label_dir / f"{img_ann.image_id}.json"
267
256
  if src_label.exists():
268
- _copy_or_move_file(
269
- src_label, part_dir, move=True, logger=self.logger
270
- )
257
+ _copy_or_move_file(src_label, part_dir, move=True, logger=self.logger)
271
258
 
272
259
  elif mode == "both":
273
260
  labels_subdir = part_dir / "labels"
@@ -282,25 +269,18 @@ class PartitionAnalyser(BaseAnalyser):
282
269
  if not wr.success:
283
270
  for err in wr.errors:
284
271
  result.add_error(
285
- f"Write {part_dir.name}/labels/"
286
- f"{img_ann.image_id}: {err}"
272
+ f"Write {part_dir.name}/labels/{img_ann.image_id}: {err}"
287
273
  )
288
274
  except Exception as e:
289
- result.add_error(
290
- f"Write {part_dir.name}/labels/"
291
- f"{img_ann.image_id}: {e}"
292
- )
275
+ result.add_error(f"Write {part_dir.name}/labels/{img_ann.image_id}: {e}")
293
276
 
294
277
  # Match and copy/move image
295
278
  stem = img_ann.image_id
296
279
  if stem in image_stems:
297
- _copy_or_move_file(
298
- image_stems[stem], images_subdir, move, self.logger
299
- )
280
+ _copy_or_move_file(image_stems[stem], images_subdir, move, self.logger)
300
281
  else:
301
282
  self._log_warning(
302
- f"No matching image found for label "
303
- f"'{img_ann.image_id}' in {image_dir}"
283
+ f"No matching image found for label '{img_ann.image_id}' in {image_dir}"
304
284
  )
305
285
 
306
286
  # Move label files if in move mode
@@ -312,21 +292,16 @@ class PartitionAnalyser(BaseAnalyser):
312
292
  src_label = label_dir / f"{img_ann.image_id}.json"
313
293
  if src_label.exists():
314
294
  _copy_or_move_file(
315
- src_label, labels_subdir,
316
- move=True, logger=self.logger
295
+ src_label, labels_subdir, move=True, logger=self.logger
317
296
  )
318
297
 
319
298
  # Report unmatched images (in image_dir but not in labels)
320
299
  if not move: # Only warn for copy mode; move mode self-resolves
321
- label_stems = {
322
- img_ann.image_id
323
- for img_ann in items[start:end]
324
- }
300
+ label_stems = {img_ann.image_id for img_ann in items[start:end]}
325
301
  for stem, img_path in image_stems.items():
326
302
  if stem not in label_stems:
327
303
  self._log_warning(
328
- f"Image '{img_path.name}' has no matching "
329
- f"label — skipped"
304
+ f"Image '{img_path.name}' has no matching label — skipped"
330
305
  )
331
306
 
332
307
  # Copy class_file to partition directory
@@ -336,10 +311,7 @@ class PartitionAnalyser(BaseAnalyser):
336
311
  if not target_cf.exists():
337
312
  shutil.copy2(str(class_file), str(target_cf))
338
313
  except OSError as e:
339
- result.add_warning(
340
- f"Could not copy class file to "
341
- f"{part_dir.name}: {e}"
342
- )
314
+ result.add_warning(f"Could not copy class file to {part_dir.name}: {e}")
343
315
 
344
316
  # ------------------------------------------------------------------
345
317
  # 6. Build result
@@ -361,8 +333,9 @@ class PartitionAnalyser(BaseAnalyser):
361
333
  # ------------------------------------------------------------------
362
334
  # 7. Log output
363
335
  # ------------------------------------------------------------------
364
- mode_label = {"images": "Images Only", "labels": "Labels Only",
365
- "both": "Labels + Images"}[mode]
336
+ mode_label = {"images": "Images Only", "labels": "Labels Only", "both": "Labels + Images"}[
337
+ mode
338
+ ]
366
339
  label_paths = label_dir if label_dir else image_dir
367
340
  self._log_info(
368
341
  format_analyse_header(
@@ -385,8 +358,6 @@ class PartitionAnalyser(BaseAnalyser):
385
358
  )
386
359
  )
387
360
  if result.log_path:
388
- self._log_info(
389
- format_analyse_result("✓ Success", result.log_path)
390
- )
361
+ self._log_info(format_analyse_result("✓ Success", result.log_path))
391
362
 
392
363
  return result
@@ -103,15 +103,11 @@ class SampleAnalyser(BaseAnalyser):
103
103
  # 1. Validate inputs
104
104
  # ------------------------------------------------------------------
105
105
  if label_dir is None and image_dir is None:
106
- result.add_error(
107
- "At least one of label_dir or image_dir must be provided"
108
- )
106
+ result.add_error("At least one of label_dir or image_dir must be provided")
109
107
  return result
110
108
 
111
109
  if count < 1:
112
- result.add_error(
113
- f"Count must be at least 1, got: {count}"
114
- )
110
+ result.add_error(f"Count must be at least 1, got: {count}")
115
111
  return result
116
112
 
117
113
  # ------------------------------------------------------------------
@@ -190,8 +186,7 @@ class SampleAnalyser(BaseAnalyser):
190
186
  actual_count = min(count, total)
191
187
  if count > total:
192
188
  result.add_warning(
193
- f"Requested {count} files but only {total} available — "
194
- f"collecting all"
189
+ f"Requested {count} files but only {total} available — collecting all"
195
190
  )
196
191
 
197
192
  sampled = items[:actual_count]
@@ -221,13 +216,10 @@ class SampleAnalyser(BaseAnalyser):
221
216
  _copy_or_move_file(src_path, label_subdir, move, self.logger)
222
217
  # Match and copy image
223
218
  if stem in image_stems:
224
- _copy_or_move_file(
225
- image_stems[stem], image_subdir, move, self.logger
226
- )
219
+ _copy_or_move_file(image_stems[stem], image_subdir, move, self.logger)
227
220
  else:
228
221
  self._log_warning(
229
- f"No matching image found for label '{stem}' "
230
- f"in image directory"
222
+ f"No matching image found for label '{stem}' in image directory"
231
223
  )
232
224
  unmatched_image_warnings += 1
233
225
  else:
@@ -238,10 +230,7 @@ class SampleAnalyser(BaseAnalyser):
238
230
  sampled_stems = {stem for _, stem in sampled}
239
231
  for stem, img_path in image_stems.items():
240
232
  if stem not in sampled_stems:
241
- self._log_warning(
242
- f"Image '{img_path.name}' has no matching "
243
- f"label — skipped"
244
- )
233
+ self._log_warning(f"Image '{img_path.name}' has no matching label — skipped")
245
234
 
246
235
  # ------------------------------------------------------------------
247
236
  # 7. Copy class_file to output_dir if provided
@@ -255,9 +244,7 @@ class SampleAnalyser(BaseAnalyser):
255
244
  else:
256
245
  shutil.copy2(str(class_file), str(target_cf))
257
246
  except OSError as e:
258
- result.add_warning(
259
- f"Could not copy class file: {e}"
260
- )
247
+ result.add_warning(f"Could not copy class file: {e}")
261
248
 
262
249
  # ------------------------------------------------------------------
263
250
  # 8. Build result
@@ -306,8 +293,6 @@ class SampleAnalyser(BaseAnalyser):
306
293
  )
307
294
  )
308
295
  if result.log_path:
309
- self._log_info(
310
- format_analyse_result("✓ Success", result.log_path)
311
- )
296
+ self._log_info(format_analyse_result("✓ Success", result.log_path))
312
297
 
313
298
  return result
@@ -101,15 +101,11 @@ class SplitAnalyser(BaseAnalyser):
101
101
  # 1. Validate inputs
102
102
  # ------------------------------------------------------------------
103
103
  if label_dir is None and image_dir is None:
104
- result.add_error(
105
- "At least one of label_dir or image_dir must be provided"
106
- )
104
+ result.add_error("At least one of label_dir or image_dir must be provided")
107
105
  return result
108
106
 
109
107
  if not 0.0 < ratio < 1.0:
110
- result.add_error(
111
- f"Ratio must be between 0 and 1 (exclusive), got: {ratio}"
112
- )
108
+ result.add_error(f"Ratio must be between 0 and 1 (exclusive), got: {ratio}")
113
109
  return result
114
110
 
115
111
  # ------------------------------------------------------------------
@@ -238,10 +234,7 @@ class SplitAnalyser(BaseAnalyser):
238
234
  label_stems = {stem for _, stem in train_items + val_items}
239
235
  for stem, img_path in image_stems.items():
240
236
  if stem not in label_stems:
241
- self._log_warning(
242
- f"Image '{img_path.name}' has no matching "
243
- f"label — skipped"
244
- )
237
+ self._log_warning(f"Image '{img_path.name}' has no matching label — skipped")
245
238
 
246
239
  # ------------------------------------------------------------------
247
240
  # 7. Copy class_file to both output dirs if provided
@@ -302,9 +295,7 @@ class SplitAnalyser(BaseAnalyser):
302
295
  )
303
296
  )
304
297
  if result.log_path:
305
- self._log_info(
306
- format_analyse_result("✓ Success", result.log_path)
307
- )
298
+ self._log_info(format_analyse_result("✓ Success", result.log_path))
308
299
 
309
300
  return result
310
301
 
@@ -343,26 +334,16 @@ def _split_files(
343
334
  for label_path, stem in train_items:
344
335
  _copy_or_move_file(label_path, train_label_dir, move, logger)
345
336
  if stem in image_stems:
346
- _copy_or_move_file(
347
- image_stems[stem], train_image_dir, move, logger
348
- )
337
+ _copy_or_move_file(image_stems[stem], train_image_dir, move, logger)
349
338
  else:
350
- logger.warning(
351
- f"No matching image found for label '{stem}' in "
352
- f"image directory"
353
- )
339
+ logger.warning(f"No matching image found for label '{stem}' in image directory")
354
340
 
355
341
  for label_path, stem in val_items:
356
342
  _copy_or_move_file(label_path, val_label_dir, move, logger)
357
343
  if stem in image_stems:
358
- _copy_or_move_file(
359
- image_stems[stem], val_image_dir, move, logger
360
- )
344
+ _copy_or_move_file(image_stems[stem], val_image_dir, move, logger)
361
345
  else:
362
- logger.warning(
363
- f"No matching image found for label '{stem}' in "
364
- f"image directory"
365
- )
346
+ logger.warning(f"No matching image found for label '{stem}' in image directory")
366
347
  else:
367
348
  train_dir = output_dir / "train"
368
349
  val_dir = output_dir / "val"