dataflow-cv 2.0.0__tar.gz → 3.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. {dataflow_cv-2.0.0/dataflow_cv.egg-info → dataflow_cv-3.0.0}/PKG-INFO +65 -34
  2. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/README.md +63 -30
  3. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/__init__.py +1 -1
  4. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/filter.py +30 -58
  5. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/log_templates.py +27 -29
  6. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/partition.py +17 -46
  7. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/sample.py +8 -23
  8. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/split.py +8 -27
  9. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/stats.py +23 -21
  10. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/utils.py +12 -31
  11. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/__init__.py +1 -1
  12. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/__init__.py +1 -1
  13. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/analyse.py +18 -34
  14. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/convert.py +28 -15
  15. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/evaluate.py +16 -4
  16. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/utils.py +12 -10
  17. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/visualize.py +15 -7
  18. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/exceptions.py +1 -1
  19. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/main.py +14 -9
  20. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/base.py +81 -73
  21. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/coco_and_labelme.py +42 -31
  22. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/labelme_and_yolo.py +30 -40
  23. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/log_templates.py +61 -1
  24. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/rle_converter.py +3 -10
  25. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/utils.py +52 -42
  26. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/yolo_and_coco.py +38 -46
  27. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/base.py +6 -16
  28. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/evaluator.py +1 -1
  29. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/log_templates.py +40 -14
  30. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/metrics.py +28 -32
  31. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/result.py +23 -26
  32. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/utils.py +21 -34
  33. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/__init__.py +8 -2
  34. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/base.py +20 -48
  35. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/coco_handler.py +65 -92
  36. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/labelme_handler.py +51 -99
  37. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/models.py +7 -21
  38. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/utils.py +48 -7
  39. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/yolo_handler.py +78 -143
  40. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/util/logging.py +8 -28
  41. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/__init__.py +1 -2
  42. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/base.py +21 -60
  43. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/coco_visualizer.py +4 -10
  44. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/labelme_visualizer.py +2 -7
  45. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/log_templates.py +1 -1
  46. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/utils.py +0 -1
  47. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/yolo_visualizer.py +4 -12
  48. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0/dataflow_cv.egg-info}/PKG-INFO +65 -34
  49. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/requires.txt +1 -3
  50. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/pyproject.toml +10 -26
  51. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/LICENSE +0 -0
  52. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/__init__.py +0 -0
  53. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/base.py +0 -0
  54. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/__init__.py +0 -0
  55. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/__init__.py +0 -0
  56. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/util/__init__.py +0 -0
  57. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/SOURCES.txt +0 -0
  58. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/dependency_links.txt +0 -0
  59. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/entry_points.txt +0 -0
  60. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/not-zip-safe +0 -0
  61. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/top_level.txt +0 -0
  62. {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dataflow-cv
3
- Version: 2.0.0
3
+ Version: 3.0.0
4
4
  Summary: A computer vision dataset processing library — analyse, convert, visualize, and evaluate annotations across YOLO, LabelMe, and COCO formats
5
5
  Author: DataFlow-CV Team
6
6
  License: MIT
@@ -33,9 +33,7 @@ Requires-Dist: pycocotools>=2.0.0; extra == "coco"
33
33
  Provides-Extra: dev
34
34
  Requires-Dist: pytest>=7.0.0; extra == "dev"
35
35
  Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
36
- Requires-Dist: black>=22.0.0; extra == "dev"
37
- Requires-Dist: isort>=5.12.0; extra == "dev"
38
- Requires-Dist: flake8>=6.0.0; extra == "dev"
36
+ Requires-Dist: ruff>=0.16; extra == "dev"
39
37
  Requires-Dist: mypy>=1.0.0; extra == "dev"
40
38
  Dynamic: license-file
41
39
 
@@ -67,6 +65,7 @@ A computer vision dataset processing library — analyse, convert, visualize, an
67
65
  | 🎨 **Visualize** | OpenCV rendering with color-coded classes, display & save modes | `dataflow-cv visualize yolo ...` |
68
66
  | 📊 **Evaluate** | COCO mAP via pycocotools, single-threshold P/R/F1 per class | `dataflow-cv evaluate detection ...` |
69
67
  | 💻 **CLI + API** | Click-based CLI with rich `--help`; Python API for pipelines | `from dataflow.convert import ...` |
68
+ | 🤖 **AI Skills** | Claude Code skill (`/dataflow:dataflow-cv`) for AI assistants — CLI/API reference & known gotchas | `claude plugin install dataflow@claude-skills` |
70
69
 
71
70
  ---
72
71
 
@@ -157,6 +156,12 @@ dataflow-cv convert yolo2coco --verbose images/ labels/ classes.txt output.json
157
156
  dataflow-cv convert yolo2coco --no-strict images/ labels/ classes.txt output.json
158
157
  ```
159
158
 
159
+ > 📂 **COCO→YOLO / COCO→LabelMe output naming**: each `.txt` / `.json` is named by the
160
+ > image file stem — leading zeros preserved (`000001.jpg` → `000001.txt`). The
161
+ > converter generates `labels/` (or the `.json` files) + `classes.txt`; the
162
+ > `images/` directory is created but left **empty** — images are never copied,
163
+ > place your image files there yourself.
164
+
160
165
  #### 🎨 Visualization
161
166
 
162
167
  ```bash
@@ -232,7 +237,13 @@ Two evaluation modes, distinguished by how overlap is measured:
232
237
 
233
238
  ```python
234
239
  from dataflow.util.logging import LogConfig
235
- from dataflow.analyse import StatsAnalyser, SplitAnalyser, FilterAnalyser, PartitionAnalyser, SampleAnalyser
240
+ from dataflow.analyse import (
241
+ StatsAnalyser,
242
+ SplitAnalyser,
243
+ FilterAnalyser,
244
+ PartitionAnalyser,
245
+ SampleAnalyser,
246
+ )
236
247
  from dataflow.convert import YoloAndCocoConverter
237
248
  from dataflow.visualize import YOLOVisualizer
238
249
  from dataflow.evaluate import DetectionEvaluator, compute_pr_f1
@@ -248,37 +259,50 @@ print(f"{result.data.total_files} images, {result.data.total_annotations} object
248
259
  # Train/test split (YOLO / LabelMe)
249
260
  splitter = SplitAnalyser(log_config=log_cfg)
250
261
  result = splitter.analyse(
251
- output_dir="output/", ratio=0.8, seed=42,
252
- label_dir="yolo_labels/", class_file="classes.txt",
262
+ output_dir="output/",
263
+ ratio=0.8,
264
+ seed=42,
265
+ label_dir="yolo_labels/",
266
+ class_file="classes.txt",
253
267
  )
254
268
  print(f"Train: {result.data.train_count}, Val: {result.data.val_count}")
255
269
 
256
270
  # Split with images (both mode — labels drive, images follow by stem)
257
271
  result = splitter.analyse(
258
- output_dir="output/", ratio=0.8, seed=42,
259
- label_dir="yolo_labels/", image_dir="images/",
272
+ output_dir="output/",
273
+ ratio=0.8,
274
+ seed=42,
275
+ label_dir="yolo_labels/",
276
+ image_dir="images/",
260
277
  class_file="classes.txt",
261
278
  )
262
279
 
263
280
  # Category filter (keep / remap categories per new classes.txt)
264
281
  filterer = FilterAnalyser(log_config=log_cfg)
265
282
  result = filterer.analyse(
266
- "yolo_labels/", original_class_file="classes.txt",
267
- new_class_file="classes_new.txt", output_dir="filtered/",
283
+ "yolo_labels/",
284
+ original_class_file="classes.txt",
285
+ new_class_file="classes_new.txt",
286
+ output_dir="filtered/",
268
287
  )
269
288
 
270
289
  # N-way partition (YOLO / LabelMe labels; images follow by stem)
271
290
  partitioner = PartitionAnalyser(log_config=log_cfg)
272
291
  result = partitioner.analyse(
273
- output_dir="parts/", num=4,
274
- label_dir="yolo_labels/", image_dir="images/",
292
+ output_dir="parts/",
293
+ num=4,
294
+ label_dir="yolo_labels/",
295
+ image_dir="images/",
275
296
  )
276
297
 
277
298
  # File sampling (labels, images, or both — random or sequential)
278
299
  sampler = SampleAnalyser(log_config=log_cfg)
279
300
  result = sampler.analyse(
280
- output_dir="sampled/", count=10,
281
- label_dir="yolo_labels/", shuffle=True, seed=42,
301
+ output_dir="sampled/",
302
+ count=10,
303
+ label_dir="yolo_labels/",
304
+ shuffle=True,
305
+ seed=42,
282
306
  )
283
307
 
284
308
  # ── Convert ──────────────────────────────────────────
@@ -286,22 +310,30 @@ result = sampler.analyse(
286
310
  log_cfg = LogConfig(name="convert", verbose=True)
287
311
  converter = YoloAndCocoConverter(source_to_target=True, log_config=log_cfg, strict_mode=True)
288
312
  result = converter.convert(
289
- source_path="yolo_labels/", target_path="anno.json",
290
- class_file="classes.txt", image_dir="images/",
313
+ source_path="yolo_labels/",
314
+ target_path="anno.json",
315
+ class_file="classes.txt",
316
+ image_dir="images/",
291
317
  )
292
318
 
293
319
  # YOLO predictions → COCO (prediction mode)
294
320
  converter = YoloAndCocoConverter(source_to_target=True, prediction=True)
295
321
  result = converter.convert(
296
- source_path="yolo_preds/", target_path="pred.json",
297
- class_file="classes.txt", image_dir="images/",
322
+ source_path="yolo_preds/",
323
+ target_path="pred.json",
324
+ class_file="classes.txt",
325
+ image_dir="images/",
298
326
  )
299
327
 
300
328
  # ── Visualize ────────────────────────────────────────
301
329
  visualizer = YOLOVisualizer(
302
- label_dir="yolo_labels/", image_dir="images/",
303
- class_file="classes.txt", is_show=True, is_save=True,
304
- output_dir="visualized/", log_config=log_cfg,
330
+ label_dir="yolo_labels/",
331
+ image_dir="images/",
332
+ class_file="classes.txt",
333
+ is_show=True,
334
+ is_save=True,
335
+ output_dir="visualized/",
336
+ log_config=log_cfg,
305
337
  )
306
338
  result = visualizer.visualize()
307
339
 
@@ -371,7 +403,7 @@ For detailed developer guidance including advanced test commands, debugging, and
371
403
 
372
404
  ### 🧪 Testing
373
405
 
374
- **561 tests, 80% code coverage (5462 statements).**
406
+ **606 tests, 80% code coverage (5532 statements).**
375
407
 
376
408
  ```bash
377
409
  pytest # All tests
@@ -385,11 +417,11 @@ pytest tests/evaluate/test_evaluator.py # Single module
385
417
 
386
418
  | Module | Coverage | Highlights |
387
419
  |--------|:--------:|------------|
388
- | `dataflow/label/` | 71% | models (84%), base (82%), utils (78%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
389
- | `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (85%), split (85%), stats (83%), filter (76%), partition (74%) |
390
- | `dataflow/convert/` | 85% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (87%), coco_and_labelme (86%), base (80%), rle (80%) |
420
+ | `dataflow/label/` | 71% | models (84%), base (82%), utils (83%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
421
+ | `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (84%), split (85%), stats (82%), filter (76%), partition (74%) |
422
+ | `dataflow/convert/` | 86% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (88%), coco_and_labelme (88%), log_templates (85%), base (81%), rle (80%) |
391
423
  | `dataflow/visualize/` | 81% | yolo_vis (100%), labelme_vis (100%), coco_vis (93%), base (76%) |
392
- | `dataflow/evaluate/` | 87% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (67%) |
424
+ | `dataflow/evaluate/` | 90% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (87%) |
393
425
  | `dataflow/cli/` | 74% | main (96%), visualize cmd (87%), utils (87%), evaluate cmd (83%), analyse cmd (65%), convert cmd (52%) |
394
426
  | `dataflow/util/` | 100% | logging (100%) |
395
427
 
@@ -398,11 +430,10 @@ pytest tests/evaluate/test_evaluator.py # Single module
398
430
  ### 🎨 Code Quality
399
431
 
400
432
  ```bash
401
- pip install -e .[dev] # Install dev dependencies
402
- black dataflow tests samples # Format
403
- isort dataflow tests samples # Sort imports
404
- mypy dataflow # Type check
405
- flake8 dataflow tests samples # Lint
433
+ pip install -e .[dev] # Install dev dependencies
434
+ ruff check dataflow tests samples # Lint
435
+ ruff format --check dataflow tests samples # Format check
436
+ mypy dataflow # Type check
406
437
  ```
407
438
 
408
439
  ### 🔗 Pre-commit Hooks (Optional)
@@ -412,7 +443,7 @@ pip install pre-commit
412
443
  pre-commit install # Install git hooks (run once)
413
444
 
414
445
  # After this, every `git commit` auto-runs:
415
- # black → isort → flake8 → whitespace checks
446
+ # ruff (lint, auto-fix) → ruff format → whitespace checks
416
447
 
417
448
  pre-commit run --all-files # Manual run against all files
418
449
  ```
@@ -428,7 +459,7 @@ dataflow/
428
459
  ├── evaluate/ # pycocotools-based metrics, log templates
429
460
  ├── util/ # Unified logging (LogManager + format helpers)
430
461
  └── cli/ # CLI entry point, commands, validation
431
- tests/ # Unit & integration tests (561 tests, conftest fixtures)
462
+ tests/ # Unit & integration tests (606 tests, conftest fixtures)
432
463
  samples/ # Python API usage examples (analyse, convert, visualize, evaluate, cli)
433
464
  assets/ # Test data (det/seg by format)
434
465
  specs/ # Canonical specifications (evaluate/ + formats/ + modules/)
@@ -26,6 +26,7 @@ A computer vision dataset processing library — analyse, convert, visualize, an
26
26
  | 🎨 **Visualize** | OpenCV rendering with color-coded classes, display & save modes | `dataflow-cv visualize yolo ...` |
27
27
  | 📊 **Evaluate** | COCO mAP via pycocotools, single-threshold P/R/F1 per class | `dataflow-cv evaluate detection ...` |
28
28
  | 💻 **CLI + API** | Click-based CLI with rich `--help`; Python API for pipelines | `from dataflow.convert import ...` |
29
+ | 🤖 **AI Skills** | Claude Code skill (`/dataflow:dataflow-cv`) for AI assistants — CLI/API reference & known gotchas | `claude plugin install dataflow@claude-skills` |
29
30
 
30
31
  ---
31
32
 
@@ -116,6 +117,12 @@ dataflow-cv convert yolo2coco --verbose images/ labels/ classes.txt output.json
116
117
  dataflow-cv convert yolo2coco --no-strict images/ labels/ classes.txt output.json
117
118
  ```
118
119
 
120
+ > 📂 **COCO→YOLO / COCO→LabelMe output naming**: each `.txt` / `.json` is named by the
121
+ > image file stem — leading zeros preserved (`000001.jpg` → `000001.txt`). The
122
+ > converter generates `labels/` (or the `.json` files) + `classes.txt`; the
123
+ > `images/` directory is created but left **empty** — images are never copied,
124
+ > place your image files there yourself.
125
+
119
126
  #### 🎨 Visualization
120
127
 
121
128
  ```bash
@@ -191,7 +198,13 @@ Two evaluation modes, distinguished by how overlap is measured:
191
198
 
192
199
  ```python
193
200
  from dataflow.util.logging import LogConfig
194
- from dataflow.analyse import StatsAnalyser, SplitAnalyser, FilterAnalyser, PartitionAnalyser, SampleAnalyser
201
+ from dataflow.analyse import (
202
+ StatsAnalyser,
203
+ SplitAnalyser,
204
+ FilterAnalyser,
205
+ PartitionAnalyser,
206
+ SampleAnalyser,
207
+ )
195
208
  from dataflow.convert import YoloAndCocoConverter
196
209
  from dataflow.visualize import YOLOVisualizer
197
210
  from dataflow.evaluate import DetectionEvaluator, compute_pr_f1
@@ -207,37 +220,50 @@ print(f"{result.data.total_files} images, {result.data.total_annotations} object
207
220
  # Train/test split (YOLO / LabelMe)
208
221
  splitter = SplitAnalyser(log_config=log_cfg)
209
222
  result = splitter.analyse(
210
- output_dir="output/", ratio=0.8, seed=42,
211
- label_dir="yolo_labels/", class_file="classes.txt",
223
+ output_dir="output/",
224
+ ratio=0.8,
225
+ seed=42,
226
+ label_dir="yolo_labels/",
227
+ class_file="classes.txt",
212
228
  )
213
229
  print(f"Train: {result.data.train_count}, Val: {result.data.val_count}")
214
230
 
215
231
  # Split with images (both mode — labels drive, images follow by stem)
216
232
  result = splitter.analyse(
217
- output_dir="output/", ratio=0.8, seed=42,
218
- label_dir="yolo_labels/", image_dir="images/",
233
+ output_dir="output/",
234
+ ratio=0.8,
235
+ seed=42,
236
+ label_dir="yolo_labels/",
237
+ image_dir="images/",
219
238
  class_file="classes.txt",
220
239
  )
221
240
 
222
241
  # Category filter (keep / remap categories per new classes.txt)
223
242
  filterer = FilterAnalyser(log_config=log_cfg)
224
243
  result = filterer.analyse(
225
- "yolo_labels/", original_class_file="classes.txt",
226
- new_class_file="classes_new.txt", output_dir="filtered/",
244
+ "yolo_labels/",
245
+ original_class_file="classes.txt",
246
+ new_class_file="classes_new.txt",
247
+ output_dir="filtered/",
227
248
  )
228
249
 
229
250
  # N-way partition (YOLO / LabelMe labels; images follow by stem)
230
251
  partitioner = PartitionAnalyser(log_config=log_cfg)
231
252
  result = partitioner.analyse(
232
- output_dir="parts/", num=4,
233
- label_dir="yolo_labels/", image_dir="images/",
253
+ output_dir="parts/",
254
+ num=4,
255
+ label_dir="yolo_labels/",
256
+ image_dir="images/",
234
257
  )
235
258
 
236
259
  # File sampling (labels, images, or both — random or sequential)
237
260
  sampler = SampleAnalyser(log_config=log_cfg)
238
261
  result = sampler.analyse(
239
- output_dir="sampled/", count=10,
240
- label_dir="yolo_labels/", shuffle=True, seed=42,
262
+ output_dir="sampled/",
263
+ count=10,
264
+ label_dir="yolo_labels/",
265
+ shuffle=True,
266
+ seed=42,
241
267
  )
242
268
 
243
269
  # ── Convert ──────────────────────────────────────────
@@ -245,22 +271,30 @@ result = sampler.analyse(
245
271
  log_cfg = LogConfig(name="convert", verbose=True)
246
272
  converter = YoloAndCocoConverter(source_to_target=True, log_config=log_cfg, strict_mode=True)
247
273
  result = converter.convert(
248
- source_path="yolo_labels/", target_path="anno.json",
249
- class_file="classes.txt", image_dir="images/",
274
+ source_path="yolo_labels/",
275
+ target_path="anno.json",
276
+ class_file="classes.txt",
277
+ image_dir="images/",
250
278
  )
251
279
 
252
280
  # YOLO predictions → COCO (prediction mode)
253
281
  converter = YoloAndCocoConverter(source_to_target=True, prediction=True)
254
282
  result = converter.convert(
255
- source_path="yolo_preds/", target_path="pred.json",
256
- class_file="classes.txt", image_dir="images/",
283
+ source_path="yolo_preds/",
284
+ target_path="pred.json",
285
+ class_file="classes.txt",
286
+ image_dir="images/",
257
287
  )
258
288
 
259
289
  # ── Visualize ────────────────────────────────────────
260
290
  visualizer = YOLOVisualizer(
261
- label_dir="yolo_labels/", image_dir="images/",
262
- class_file="classes.txt", is_show=True, is_save=True,
263
- output_dir="visualized/", log_config=log_cfg,
291
+ label_dir="yolo_labels/",
292
+ image_dir="images/",
293
+ class_file="classes.txt",
294
+ is_show=True,
295
+ is_save=True,
296
+ output_dir="visualized/",
297
+ log_config=log_cfg,
264
298
  )
265
299
  result = visualizer.visualize()
266
300
 
@@ -330,7 +364,7 @@ For detailed developer guidance including advanced test commands, debugging, and
330
364
 
331
365
  ### 🧪 Testing
332
366
 
333
- **561 tests, 80% code coverage (5462 statements).**
367
+ **606 tests, 80% code coverage (5532 statements).**
334
368
 
335
369
  ```bash
336
370
  pytest # All tests
@@ -344,11 +378,11 @@ pytest tests/evaluate/test_evaluator.py # Single module
344
378
 
345
379
  | Module | Coverage | Highlights |
346
380
  |--------|:--------:|------------|
347
- | `dataflow/label/` | 71% | models (84%), base (82%), utils (78%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
348
- | `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (85%), split (85%), stats (83%), filter (76%), partition (74%) |
349
- | `dataflow/convert/` | 85% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (87%), coco_and_labelme (86%), base (80%), rle (80%) |
381
+ | `dataflow/label/` | 71% | models (84%), base (82%), utils (83%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
382
+ | `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (84%), split (85%), stats (82%), filter (76%), partition (74%) |
383
+ | `dataflow/convert/` | 86% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (88%), coco_and_labelme (88%), log_templates (85%), base (81%), rle (80%) |
350
384
  | `dataflow/visualize/` | 81% | yolo_vis (100%), labelme_vis (100%), coco_vis (93%), base (76%) |
351
- | `dataflow/evaluate/` | 87% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (67%) |
385
+ | `dataflow/evaluate/` | 90% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (87%) |
352
386
  | `dataflow/cli/` | 74% | main (96%), visualize cmd (87%), utils (87%), evaluate cmd (83%), analyse cmd (65%), convert cmd (52%) |
353
387
  | `dataflow/util/` | 100% | logging (100%) |
354
388
 
@@ -357,11 +391,10 @@ pytest tests/evaluate/test_evaluator.py # Single module
357
391
  ### 🎨 Code Quality
358
392
 
359
393
  ```bash
360
- pip install -e .[dev] # Install dev dependencies
361
- black dataflow tests samples # Format
362
- isort dataflow tests samples # Sort imports
363
- mypy dataflow # Type check
364
- flake8 dataflow tests samples # Lint
394
+ pip install -e .[dev] # Install dev dependencies
395
+ ruff check dataflow tests samples # Lint
396
+ ruff format --check dataflow tests samples # Format check
397
+ mypy dataflow # Type check
365
398
  ```
366
399
 
367
400
  ### 🔗 Pre-commit Hooks (Optional)
@@ -371,7 +404,7 @@ pip install pre-commit
371
404
  pre-commit install # Install git hooks (run once)
372
405
 
373
406
  # After this, every `git commit` auto-runs:
374
- # black → isort → flake8 → whitespace checks
407
+ # ruff (lint, auto-fix) → ruff format → whitespace checks
375
408
 
376
409
  pre-commit run --all-files # Manual run against all files
377
410
  ```
@@ -387,7 +420,7 @@ dataflow/
387
420
  ├── evaluate/ # pycocotools-based metrics, log templates
388
421
  ├── util/ # Unified logging (LogManager + format helpers)
389
422
  └── cli/ # CLI entry point, commands, validation
390
- tests/ # Unit & integration tests (561 tests, conftest fixtures)
423
+ tests/ # Unit & integration tests (606 tests, conftest fixtures)
391
424
  samples/ # Python API usage examples (analyse, convert, visualize, evaluate, cli)
392
425
  assets/ # Test data (det/seg by format)
393
426
  specs/ # Canonical specifications (evaluate/ + formats/ + modules/)
@@ -1,4 +1,4 @@
1
1
  """DataFlow-CV: A computer vision dataset processing library."""
2
2
 
3
- __version__ = "2.0.0"
3
+ __version__ = "3.0.0"
4
4
  __author__ = "DataFlow-CV Team"
@@ -24,8 +24,7 @@ from .log_templates import (
24
24
  format_filter_result,
25
25
  )
26
26
  from .utils import create_handler, detect_format, load_class_names
27
- from dataflow.label.models import (DatasetAnnotations, ImageAnnotation,
28
- ObjectAnnotation)
27
+ from dataflow.label.models import DatasetAnnotations, ImageAnnotation, ObjectAnnotation
29
28
 
30
29
 
31
30
  class FilterAnalyser(BaseAnalyser):
@@ -99,17 +98,14 @@ class FilterAnalyser(BaseAnalyser):
99
98
  for new_id, name in new_classes.items():
100
99
  if name in name_to_old_id:
101
100
  old_id = name_to_old_id[name]
102
- mapping = CategoryMapping(
103
- new_id=new_id, old_id=old_id, name=name
104
- )
101
+ mapping = CategoryMapping(new_id=new_id, old_id=old_id, name=name)
105
102
  old_to_new[old_id] = mapping
106
103
  kept.append(mapping)
107
104
  else:
108
105
  missing.append(name)
109
106
  if logger:
110
107
  logger.warning(
111
- f'Category "{name}" in new class file not '
112
- f"found in source — skipping"
108
+ f'Category "{name}" in new class file not found in source — skipping'
113
109
  )
114
110
 
115
111
  # Build removed list
@@ -134,10 +130,7 @@ class FilterAnalyser(BaseAnalyser):
134
130
  total_after = 0
135
131
 
136
132
  for image_ann in dataset.images:
137
- filtered = [
138
- obj for obj in image_ann.objects
139
- if obj.class_id in old_to_new
140
- ]
133
+ filtered = [obj for obj in image_ann.objects if obj.class_id in old_to_new]
141
134
  for obj in filtered:
142
135
  mapping = old_to_new[obj.class_id]
143
136
  obj.class_id = mapping.new_id
@@ -202,9 +195,7 @@ class FilterAnalyser(BaseAnalyser):
202
195
  return result
203
196
 
204
197
  if not new_classes:
205
- result.add_error(
206
- f"No valid class names in new class file: {new_class_file}"
207
- )
198
+ result.add_error(f"No valid class names in new class file: {new_class_file}")
208
199
  return result
209
200
 
210
201
  # ---- 3. Detect format + create handler ------------------------
@@ -259,9 +250,7 @@ class FilterAnalyser(BaseAnalyser):
259
250
  )
260
251
 
261
252
  if not old_to_new:
262
- result.add_error(
263
- "No matching categories between source and new class file"
264
- )
253
+ result.add_error("No matching categories between source and new class file")
265
254
  return result
266
255
 
267
256
  # ---- 5. Ensure output directory --------------------------------
@@ -304,14 +293,16 @@ class FilterAnalyser(BaseAnalyser):
304
293
  # rather than mutating the original (avoid
305
294
  # aliasing — the original may be reused by the
306
295
  # iterator or shared across images).
307
- filtered_objects.append(ObjectAnnotation(
308
- class_id=mapping.new_id,
309
- class_name=mapping.name,
310
- bbox=obj.bbox,
311
- segmentation=obj.segmentation,
312
- confidence=obj.confidence,
313
- is_crowd=obj.is_crowd,
314
- ))
296
+ filtered_objects.append(
297
+ ObjectAnnotation(
298
+ class_id=mapping.new_id,
299
+ class_name=mapping.name,
300
+ bbox=obj.bbox,
301
+ segmentation=obj.segmentation,
302
+ confidence=obj.confidence,
303
+ is_crowd=obj.is_crowd,
304
+ )
305
+ )
315
306
  total_after += len(filtered_objects)
316
307
 
317
308
  if filtered_objects:
@@ -328,9 +319,7 @@ class FilterAnalyser(BaseAnalyser):
328
319
  wr = write_handler.write_one(filtered_img, output_dir)
329
320
  if not wr.success:
330
321
  for err in wr.errors:
331
- result.add_error(
332
- f"Write {image_ann.image_id}: {err}"
333
- )
322
+ result.add_error(f"Write {image_ann.image_id}: {err}")
334
323
  return result
335
324
  except Exception as e:
336
325
  result.add_error(f"Failed during streaming filter: {e}")
@@ -342,23 +331,19 @@ class FilterAnalyser(BaseAnalyser):
342
331
  total_files = dataset.num_images
343
332
  total_before = dataset.num_objects
344
333
 
345
- total_files_with_annotations, total_after = (
346
- self._filter_dataset_images(dataset, old_to_new)
334
+ total_files_with_annotations, total_after = self._filter_dataset_images(
335
+ dataset, old_to_new
347
336
  )
348
337
 
349
338
  # Update categories
350
- dataset.categories = {
351
- km.new_id: km.name for km in kept_categories
352
- }
339
+ dataset.categories = {km.new_id: km.name for km in kept_categories}
353
340
 
354
341
  try:
355
342
  for image_ann in dataset.images:
356
343
  wr = handler.write_one(image_ann, output_dir)
357
344
  if not wr.success:
358
345
  for err in wr.errors:
359
- result.add_error(
360
- f"Write {image_ann.image_id}: {err}"
361
- )
346
+ result.add_error(f"Write {image_ann.image_id}: {err}")
362
347
  return result
363
348
  except Exception as e:
364
349
  result.add_error(f"Failed to write filtered output: {e}")
@@ -369,14 +354,12 @@ class FilterAnalyser(BaseAnalyser):
369
354
  total_files = dataset.num_images
370
355
  total_before = dataset.num_objects
371
356
 
372
- total_files_with_annotations, total_after = (
373
- self._filter_dataset_images(dataset, old_to_new)
357
+ total_files_with_annotations, total_after = self._filter_dataset_images(
358
+ dataset, old_to_new
374
359
  )
375
360
 
376
361
  # Update categories to match new class file
377
- dataset.categories = {
378
- km.new_id: km.name for km in kept_categories
379
- }
362
+ dataset.categories = {km.new_id: km.name for km in kept_categories}
380
363
 
381
364
  output_path = output_dir / label_path.name
382
365
  try:
@@ -413,29 +396,20 @@ class FilterAnalyser(BaseAnalyser):
413
396
  if missing_categories:
414
397
  s = "y" if len(missing_categories) == 1 else "ies"
415
398
  result.add_warning(
416
- f"{len(missing_categories)} categor{s} in new class "
417
- f"file not found in source"
399
+ f"{len(missing_categories)} categor{s} in new class file not found in source"
418
400
  )
419
401
 
420
402
  if total_after == 0 and total_before > 0:
421
- result.add_warning(
422
- "All annotations were filtered out — output files are empty"
423
- )
403
+ result.add_warning("All annotations were filtered out — output files are empty")
424
404
 
425
405
  # ---- 9. Log output ---------------------------------------------
426
406
  self._log_info(
427
- format_analyse_header(
428
- "Category Filter", label_path, f"{fmt} (auto-detected)"
429
- )
407
+ format_analyse_header("Category Filter", label_path, f"{fmt} (auto-detected)")
430
408
  )
431
409
  self._log_info(
432
- f" Original class: {original_class_file.name} "
433
- f"({len(original_classes)} categories)"
434
- )
435
- self._log_info(
436
- f" New class: {new_class_file.name} "
437
- f"({len(new_classes)} categories)\n"
410
+ f" Original class: {original_class_file.name} ({len(original_classes)} categories)"
438
411
  )
412
+ self._log_info(f" New class: {new_class_file.name} ({len(new_classes)} categories)\n")
439
413
  self._log_info(
440
414
  format_filter_result(
441
415
  total_files=total_files,
@@ -449,8 +423,6 @@ class FilterAnalyser(BaseAnalyser):
449
423
  )
450
424
  )
451
425
  if result.log_path:
452
- self._log_info(
453
- format_analyse_result("✓ Success", result.log_path)
454
- )
426
+ self._log_info(format_analyse_result("✓ Success", result.log_path))
455
427
 
456
428
  return result