dataflow-cv 2.0.1__tar.gz → 3.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. {dataflow_cv-2.0.1/dataflow_cv.egg-info → dataflow_cv-3.0.0}/PKG-INFO +63 -31
  2. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/README.md +62 -30
  3. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/__init__.py +1 -1
  4. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/base.py +34 -1
  5. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/coco_and_labelme.py +11 -1
  6. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/log_templates.py +61 -1
  7. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/utils.py +36 -0
  8. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/yolo_and_coco.py +5 -2
  9. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/utils.py +9 -3
  10. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/coco_handler.py +17 -10
  11. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/utils.py +46 -1
  12. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0/dataflow_cv.egg-info}/PKG-INFO +63 -31
  13. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/pyproject.toml +1 -1
  14. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/LICENSE +0 -0
  15. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/__init__.py +0 -0
  16. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/base.py +0 -0
  17. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/filter.py +0 -0
  18. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/log_templates.py +0 -0
  19. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/partition.py +0 -0
  20. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/sample.py +0 -0
  21. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/split.py +0 -0
  22. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/stats.py +0 -0
  23. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/utils.py +0 -0
  24. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/__init__.py +0 -0
  25. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/__init__.py +0 -0
  26. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/analyse.py +0 -0
  27. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/convert.py +0 -0
  28. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/evaluate.py +0 -0
  29. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/utils.py +0 -0
  30. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/visualize.py +0 -0
  31. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/exceptions.py +0 -0
  32. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/main.py +0 -0
  33. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/__init__.py +0 -0
  34. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/labelme_and_yolo.py +0 -0
  35. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/rle_converter.py +0 -0
  36. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/__init__.py +0 -0
  37. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/base.py +0 -0
  38. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/evaluator.py +0 -0
  39. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/log_templates.py +0 -0
  40. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/metrics.py +0 -0
  41. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/result.py +0 -0
  42. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/__init__.py +0 -0
  43. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/base.py +0 -0
  44. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/labelme_handler.py +0 -0
  45. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/models.py +0 -0
  46. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/yolo_handler.py +0 -0
  47. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/util/__init__.py +0 -0
  48. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/util/logging.py +0 -0
  49. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/__init__.py +0 -0
  50. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/base.py +0 -0
  51. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/coco_visualizer.py +0 -0
  52. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/labelme_visualizer.py +0 -0
  53. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/log_templates.py +0 -0
  54. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/utils.py +0 -0
  55. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/yolo_visualizer.py +0 -0
  56. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/SOURCES.txt +0 -0
  57. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/dependency_links.txt +0 -0
  58. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/entry_points.txt +0 -0
  59. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/not-zip-safe +0 -0
  60. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/requires.txt +0 -0
  61. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/top_level.txt +0 -0
  62. {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dataflow-cv
3
- Version: 2.0.1
3
+ Version: 3.0.0
4
4
  Summary: A computer vision dataset processing library — analyse, convert, visualize, and evaluate annotations across YOLO, LabelMe, and COCO formats
5
5
  Author: DataFlow-CV Team
6
6
  License: MIT
@@ -156,6 +156,12 @@ dataflow-cv convert yolo2coco --verbose images/ labels/ classes.txt output.json
156
156
  dataflow-cv convert yolo2coco --no-strict images/ labels/ classes.txt output.json
157
157
  ```
158
158
 
159
+ > 📂 **COCO→YOLO / COCO→LabelMe output naming**: each `.txt` / `.json` is named by the
160
+ > image file stem — leading zeros preserved (`000001.jpg` → `000001.txt`). The
161
+ > converter generates `labels/` (or the `.json` files) + `classes.txt`; the
162
+ > `images/` directory is created but left **empty** — images are never copied,
163
+ > place your image files there yourself.
164
+
159
165
  #### 🎨 Visualization
160
166
 
161
167
  ```bash
@@ -231,7 +237,13 @@ Two evaluation modes, distinguished by how overlap is measured:
231
237
 
232
238
  ```python
233
239
  from dataflow.util.logging import LogConfig
234
- from dataflow.analyse import StatsAnalyser, SplitAnalyser, FilterAnalyser, PartitionAnalyser, SampleAnalyser
240
+ from dataflow.analyse import (
241
+ StatsAnalyser,
242
+ SplitAnalyser,
243
+ FilterAnalyser,
244
+ PartitionAnalyser,
245
+ SampleAnalyser,
246
+ )
235
247
  from dataflow.convert import YoloAndCocoConverter
236
248
  from dataflow.visualize import YOLOVisualizer
237
249
  from dataflow.evaluate import DetectionEvaluator, compute_pr_f1
@@ -247,37 +259,50 @@ print(f"{result.data.total_files} images, {result.data.total_annotations} object
247
259
  # Train/test split (YOLO / LabelMe)
248
260
  splitter = SplitAnalyser(log_config=log_cfg)
249
261
  result = splitter.analyse(
250
- output_dir="output/", ratio=0.8, seed=42,
251
- label_dir="yolo_labels/", class_file="classes.txt",
262
+ output_dir="output/",
263
+ ratio=0.8,
264
+ seed=42,
265
+ label_dir="yolo_labels/",
266
+ class_file="classes.txt",
252
267
  )
253
268
  print(f"Train: {result.data.train_count}, Val: {result.data.val_count}")
254
269
 
255
270
  # Split with images (both mode — labels drive, images follow by stem)
256
271
  result = splitter.analyse(
257
- output_dir="output/", ratio=0.8, seed=42,
258
- label_dir="yolo_labels/", image_dir="images/",
272
+ output_dir="output/",
273
+ ratio=0.8,
274
+ seed=42,
275
+ label_dir="yolo_labels/",
276
+ image_dir="images/",
259
277
  class_file="classes.txt",
260
278
  )
261
279
 
262
280
  # Category filter (keep / remap categories per new classes.txt)
263
281
  filterer = FilterAnalyser(log_config=log_cfg)
264
282
  result = filterer.analyse(
265
- "yolo_labels/", original_class_file="classes.txt",
266
- new_class_file="classes_new.txt", output_dir="filtered/",
283
+ "yolo_labels/",
284
+ original_class_file="classes.txt",
285
+ new_class_file="classes_new.txt",
286
+ output_dir="filtered/",
267
287
  )
268
288
 
269
289
  # N-way partition (YOLO / LabelMe labels; images follow by stem)
270
290
  partitioner = PartitionAnalyser(log_config=log_cfg)
271
291
  result = partitioner.analyse(
272
- output_dir="parts/", num=4,
273
- label_dir="yolo_labels/", image_dir="images/",
292
+ output_dir="parts/",
293
+ num=4,
294
+ label_dir="yolo_labels/",
295
+ image_dir="images/",
274
296
  )
275
297
 
276
298
  # File sampling (labels, images, or both — random or sequential)
277
299
  sampler = SampleAnalyser(log_config=log_cfg)
278
300
  result = sampler.analyse(
279
- output_dir="sampled/", count=10,
280
- label_dir="yolo_labels/", shuffle=True, seed=42,
301
+ output_dir="sampled/",
302
+ count=10,
303
+ label_dir="yolo_labels/",
304
+ shuffle=True,
305
+ seed=42,
281
306
  )
282
307
 
283
308
  # ── Convert ──────────────────────────────────────────
@@ -285,22 +310,30 @@ result = sampler.analyse(
285
310
  log_cfg = LogConfig(name="convert", verbose=True)
286
311
  converter = YoloAndCocoConverter(source_to_target=True, log_config=log_cfg, strict_mode=True)
287
312
  result = converter.convert(
288
- source_path="yolo_labels/", target_path="anno.json",
289
- class_file="classes.txt", image_dir="images/",
313
+ source_path="yolo_labels/",
314
+ target_path="anno.json",
315
+ class_file="classes.txt",
316
+ image_dir="images/",
290
317
  )
291
318
 
292
319
  # YOLO predictions → COCO (prediction mode)
293
320
  converter = YoloAndCocoConverter(source_to_target=True, prediction=True)
294
321
  result = converter.convert(
295
- source_path="yolo_preds/", target_path="pred.json",
296
- class_file="classes.txt", image_dir="images/",
322
+ source_path="yolo_preds/",
323
+ target_path="pred.json",
324
+ class_file="classes.txt",
325
+ image_dir="images/",
297
326
  )
298
327
 
299
328
  # ── Visualize ────────────────────────────────────────
300
329
  visualizer = YOLOVisualizer(
301
- label_dir="yolo_labels/", image_dir="images/",
302
- class_file="classes.txt", is_show=True, is_save=True,
303
- output_dir="visualized/", log_config=log_cfg,
330
+ label_dir="yolo_labels/",
331
+ image_dir="images/",
332
+ class_file="classes.txt",
333
+ is_show=True,
334
+ is_save=True,
335
+ output_dir="visualized/",
336
+ log_config=log_cfg,
304
337
  )
305
338
  result = visualizer.visualize()
306
339
 
@@ -370,7 +403,7 @@ For detailed developer guidance including advanced test commands, debugging, and
370
403
 
371
404
  ### 🧪 Testing
372
405
 
373
- **561 tests, 80% code coverage (5462 statements).**
406
+ **606 tests, 80% code coverage (5532 statements).**
374
407
 
375
408
  ```bash
376
409
  pytest # All tests
@@ -384,11 +417,11 @@ pytest tests/evaluate/test_evaluator.py # Single module
384
417
 
385
418
  | Module | Coverage | Highlights |
386
419
  |--------|:--------:|------------|
387
- | `dataflow/label/` | 71% | models (84%), base (82%), utils (78%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
388
- | `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (85%), split (85%), stats (83%), filter (76%), partition (74%) |
389
- | `dataflow/convert/` | 85% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (87%), coco_and_labelme (86%), base (80%), rle (80%) |
420
+ | `dataflow/label/` | 71% | models (84%), base (82%), utils (83%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
421
+ | `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (84%), split (85%), stats (82%), filter (76%), partition (74%) |
422
+ | `dataflow/convert/` | 86% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (88%), coco_and_labelme (88%), log_templates (85%), base (81%), rle (80%) |
390
423
  | `dataflow/visualize/` | 81% | yolo_vis (100%), labelme_vis (100%), coco_vis (93%), base (76%) |
391
- | `dataflow/evaluate/` | 87% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (67%) |
424
+ | `dataflow/evaluate/` | 90% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (87%) |
392
425
  | `dataflow/cli/` | 74% | main (96%), visualize cmd (87%), utils (87%), evaluate cmd (83%), analyse cmd (65%), convert cmd (52%) |
393
426
  | `dataflow/util/` | 100% | logging (100%) |
394
427
 
@@ -397,11 +430,10 @@ pytest tests/evaluate/test_evaluator.py # Single module
397
430
  ### 🎨 Code Quality
398
431
 
399
432
  ```bash
400
- pip install -e .[dev] # Install dev dependencies
401
- black dataflow tests samples # Format
402
- isort dataflow tests samples # Sort imports
403
- mypy dataflow # Type check
404
- flake8 dataflow tests samples # Lint
433
+ pip install -e .[dev] # Install dev dependencies
434
+ ruff check dataflow tests samples # Lint
435
+ ruff format --check dataflow tests samples # Format check
436
+ mypy dataflow # Type check
405
437
  ```
406
438
 
407
439
  ### 🔗 Pre-commit Hooks (Optional)
@@ -411,7 +443,7 @@ pip install pre-commit
411
443
  pre-commit install # Install git hooks (run once)
412
444
 
413
445
  # After this, every `git commit` auto-runs:
414
- # black → isort → flake8 → whitespace checks
446
+ # ruff (lint, auto-fix) → ruff format → whitespace checks
415
447
 
416
448
  pre-commit run --all-files # Manual run against all files
417
449
  ```
@@ -427,7 +459,7 @@ dataflow/
427
459
  ├── evaluate/ # pycocotools-based metrics, log templates
428
460
  ├── util/ # Unified logging (LogManager + format helpers)
429
461
  └── cli/ # CLI entry point, commands, validation
430
- tests/ # Unit & integration tests (561 tests, conftest fixtures)
462
+ tests/ # Unit & integration tests (606 tests, conftest fixtures)
431
463
  samples/ # Python API usage examples (analyse, convert, visualize, evaluate, cli)
432
464
  assets/ # Test data (det/seg by format)
433
465
  specs/ # Canonical specifications (evaluate/ + formats/ + modules/)
@@ -117,6 +117,12 @@ dataflow-cv convert yolo2coco --verbose images/ labels/ classes.txt output.json
117
117
  dataflow-cv convert yolo2coco --no-strict images/ labels/ classes.txt output.json
118
118
  ```
119
119
 
120
+ > 📂 **COCO→YOLO / COCO→LabelMe output naming**: each `.txt` / `.json` is named by the
121
+ > image file stem — leading zeros preserved (`000001.jpg` → `000001.txt`). The
122
+ > converter generates `labels/` (or the `.json` files) + `classes.txt`; the
123
+ > `images/` directory is created but left **empty** — images are never copied,
124
+ > place your image files there yourself.
125
+
120
126
  #### 🎨 Visualization
121
127
 
122
128
  ```bash
@@ -192,7 +198,13 @@ Two evaluation modes, distinguished by how overlap is measured:
192
198
 
193
199
  ```python
194
200
  from dataflow.util.logging import LogConfig
195
- from dataflow.analyse import StatsAnalyser, SplitAnalyser, FilterAnalyser, PartitionAnalyser, SampleAnalyser
201
+ from dataflow.analyse import (
202
+ StatsAnalyser,
203
+ SplitAnalyser,
204
+ FilterAnalyser,
205
+ PartitionAnalyser,
206
+ SampleAnalyser,
207
+ )
196
208
  from dataflow.convert import YoloAndCocoConverter
197
209
  from dataflow.visualize import YOLOVisualizer
198
210
  from dataflow.evaluate import DetectionEvaluator, compute_pr_f1
@@ -208,37 +220,50 @@ print(f"{result.data.total_files} images, {result.data.total_annotations} object
208
220
  # Train/test split (YOLO / LabelMe)
209
221
  splitter = SplitAnalyser(log_config=log_cfg)
210
222
  result = splitter.analyse(
211
- output_dir="output/", ratio=0.8, seed=42,
212
- label_dir="yolo_labels/", class_file="classes.txt",
223
+ output_dir="output/",
224
+ ratio=0.8,
225
+ seed=42,
226
+ label_dir="yolo_labels/",
227
+ class_file="classes.txt",
213
228
  )
214
229
  print(f"Train: {result.data.train_count}, Val: {result.data.val_count}")
215
230
 
216
231
  # Split with images (both mode — labels drive, images follow by stem)
217
232
  result = splitter.analyse(
218
- output_dir="output/", ratio=0.8, seed=42,
219
- label_dir="yolo_labels/", image_dir="images/",
233
+ output_dir="output/",
234
+ ratio=0.8,
235
+ seed=42,
236
+ label_dir="yolo_labels/",
237
+ image_dir="images/",
220
238
  class_file="classes.txt",
221
239
  )
222
240
 
223
241
  # Category filter (keep / remap categories per new classes.txt)
224
242
  filterer = FilterAnalyser(log_config=log_cfg)
225
243
  result = filterer.analyse(
226
- "yolo_labels/", original_class_file="classes.txt",
227
- new_class_file="classes_new.txt", output_dir="filtered/",
244
+ "yolo_labels/",
245
+ original_class_file="classes.txt",
246
+ new_class_file="classes_new.txt",
247
+ output_dir="filtered/",
228
248
  )
229
249
 
230
250
  # N-way partition (YOLO / LabelMe labels; images follow by stem)
231
251
  partitioner = PartitionAnalyser(log_config=log_cfg)
232
252
  result = partitioner.analyse(
233
- output_dir="parts/", num=4,
234
- label_dir="yolo_labels/", image_dir="images/",
253
+ output_dir="parts/",
254
+ num=4,
255
+ label_dir="yolo_labels/",
256
+ image_dir="images/",
235
257
  )
236
258
 
237
259
  # File sampling (labels, images, or both — random or sequential)
238
260
  sampler = SampleAnalyser(log_config=log_cfg)
239
261
  result = sampler.analyse(
240
- output_dir="sampled/", count=10,
241
- label_dir="yolo_labels/", shuffle=True, seed=42,
262
+ output_dir="sampled/",
263
+ count=10,
264
+ label_dir="yolo_labels/",
265
+ shuffle=True,
266
+ seed=42,
242
267
  )
243
268
 
244
269
  # ── Convert ──────────────────────────────────────────
@@ -246,22 +271,30 @@ result = sampler.analyse(
246
271
  log_cfg = LogConfig(name="convert", verbose=True)
247
272
  converter = YoloAndCocoConverter(source_to_target=True, log_config=log_cfg, strict_mode=True)
248
273
  result = converter.convert(
249
- source_path="yolo_labels/", target_path="anno.json",
250
- class_file="classes.txt", image_dir="images/",
274
+ source_path="yolo_labels/",
275
+ target_path="anno.json",
276
+ class_file="classes.txt",
277
+ image_dir="images/",
251
278
  )
252
279
 
253
280
  # YOLO predictions → COCO (prediction mode)
254
281
  converter = YoloAndCocoConverter(source_to_target=True, prediction=True)
255
282
  result = converter.convert(
256
- source_path="yolo_preds/", target_path="pred.json",
257
- class_file="classes.txt", image_dir="images/",
283
+ source_path="yolo_preds/",
284
+ target_path="pred.json",
285
+ class_file="classes.txt",
286
+ image_dir="images/",
258
287
  )
259
288
 
260
289
  # ── Visualize ────────────────────────────────────────
261
290
  visualizer = YOLOVisualizer(
262
- label_dir="yolo_labels/", image_dir="images/",
263
- class_file="classes.txt", is_show=True, is_save=True,
264
- output_dir="visualized/", log_config=log_cfg,
291
+ label_dir="yolo_labels/",
292
+ image_dir="images/",
293
+ class_file="classes.txt",
294
+ is_show=True,
295
+ is_save=True,
296
+ output_dir="visualized/",
297
+ log_config=log_cfg,
265
298
  )
266
299
  result = visualizer.visualize()
267
300
 
@@ -331,7 +364,7 @@ For detailed developer guidance including advanced test commands, debugging, and
331
364
 
332
365
  ### 🧪 Testing
333
366
 
334
- **561 tests, 80% code coverage (5462 statements).**
367
+ **606 tests, 80% code coverage (5532 statements).**
335
368
 
336
369
  ```bash
337
370
  pytest # All tests
@@ -345,11 +378,11 @@ pytest tests/evaluate/test_evaluator.py # Single module
345
378
 
346
379
  | Module | Coverage | Highlights |
347
380
  |--------|:--------:|------------|
348
- | `dataflow/label/` | 71% | models (84%), base (82%), utils (78%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
349
- | `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (85%), split (85%), stats (83%), filter (76%), partition (74%) |
350
- | `dataflow/convert/` | 85% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (87%), coco_and_labelme (86%), base (80%), rle (80%) |
381
+ | `dataflow/label/` | 71% | models (84%), base (82%), utils (83%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
382
+ | `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (84%), split (85%), stats (82%), filter (76%), partition (74%) |
383
+ | `dataflow/convert/` | 86% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (88%), coco_and_labelme (88%), log_templates (85%), base (81%), rle (80%) |
351
384
  | `dataflow/visualize/` | 81% | yolo_vis (100%), labelme_vis (100%), coco_vis (93%), base (76%) |
352
- | `dataflow/evaluate/` | 87% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (67%) |
385
+ | `dataflow/evaluate/` | 90% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (87%) |
353
386
  | `dataflow/cli/` | 74% | main (96%), visualize cmd (87%), utils (87%), evaluate cmd (83%), analyse cmd (65%), convert cmd (52%) |
354
387
  | `dataflow/util/` | 100% | logging (100%) |
355
388
 
@@ -358,11 +391,10 @@ pytest tests/evaluate/test_evaluator.py # Single module
358
391
  ### 🎨 Code Quality
359
392
 
360
393
  ```bash
361
- pip install -e .[dev] # Install dev dependencies
362
- black dataflow tests samples # Format
363
- isort dataflow tests samples # Sort imports
364
- mypy dataflow # Type check
365
- flake8 dataflow tests samples # Lint
394
+ pip install -e .[dev] # Install dev dependencies
395
+ ruff check dataflow tests samples # Lint
396
+ ruff format --check dataflow tests samples # Format check
397
+ mypy dataflow # Type check
366
398
  ```
367
399
 
368
400
  ### 🔗 Pre-commit Hooks (Optional)
@@ -372,7 +404,7 @@ pip install pre-commit
372
404
  pre-commit install # Install git hooks (run once)
373
405
 
374
406
  # After this, every `git commit` auto-runs:
375
- # black → isort → flake8 → whitespace checks
407
+ # ruff (lint, auto-fix) → ruff format → whitespace checks
376
408
 
377
409
  pre-commit run --all-files # Manual run against all files
378
410
  ```
@@ -388,7 +420,7 @@ dataflow/
388
420
  ├── evaluate/ # pycocotools-based metrics, log templates
389
421
  ├── util/ # Unified logging (LogManager + format helpers)
390
422
  └── cli/ # CLI entry point, commands, validation
391
- tests/ # Unit & integration tests (561 tests, conftest fixtures)
423
+ tests/ # Unit & integration tests (606 tests, conftest fixtures)
392
424
  samples/ # Python API usage examples (analyse, convert, visualize, evaluate, cli)
393
425
  assets/ # Test data (det/seg by format)
394
426
  specs/ # Canonical specifications (evaluate/ + formats/ + modules/)
@@ -1,4 +1,4 @@
1
1
  """DataFlow-CV: A computer vision dataset processing library."""
2
2
 
3
- __version__ = "2.0.1"
3
+ __version__ = "3.0.0"
4
4
  __author__ = "DataFlow-CV Team"
@@ -8,7 +8,7 @@ import datetime
8
8
  from abc import ABC, abstractmethod
9
9
  from dataclasses import dataclass, field
10
10
  from pathlib import Path
11
- from typing import Any, Dict, List, Optional
11
+ from typing import Any, Dict, List, Optional, Set
12
12
 
13
13
  from ..label.base import AnnotationResult, BaseAnnotationHandler
14
14
  from ..label.models import AnnotationFormat, DatasetAnnotations, ImageAnnotation
@@ -215,6 +215,7 @@ class BaseConverter(ABC):
215
215
 
216
216
  num_images = 0
217
217
  num_objects = 0
218
+ written_ids: Set[str] = set() # Detect duplicate output stems
218
219
 
219
220
  try:
220
221
  # 1. Validate inputs
@@ -238,6 +239,38 @@ class BaseConverter(ABC):
238
239
 
239
240
  for image_ann in source_handler.iter_images():
240
241
  target_ann = self._convert_single_image(image_ann, **kwargs)
242
+
243
+ # Output-name notices: COCO-source converters rewrite
244
+ # image_id to the file_name stem (e.g. "1" → "000001"), and
245
+ # file_name directory components are dropped (per-file
246
+ # outputs are flat per the YOLO/LabelMe format specs).
247
+ renamed = target_ann.image_id != image_ann.image_id
248
+ flattened = self.source_format == "coco" and (
249
+ Path(str(image_ann.image_path).replace("\\", "/")).parent != Path(".")
250
+ )
251
+ if renamed or flattened:
252
+ parts = []
253
+ if renamed:
254
+ parts.append(f"renamed {image_ann.image_id} → {target_ann.image_id}")
255
+ if flattened:
256
+ parts.append(f"subdirectory flattened from '{image_ann.image_path}'")
257
+ notice = f"Output '{target_ann.image_id}': " + ", ".join(parts)
258
+ result.add_warning(notice)
259
+ self._log_warning(notice)
260
+
261
+ # Duplicate output stem detection: two source images sharing
262
+ # a basename (e.g. in different subdirectories) would silently
263
+ # overwrite each other's output file.
264
+ if target_ann.image_id in written_ids:
265
+ dup = (
266
+ f"Duplicate output stem '{target_ann.image_id}': "
267
+ f"{image_ann.image_path} overwrites a previously written file"
268
+ )
269
+ result.add_warning(dup)
270
+ self._log_warning(dup)
271
+ else:
272
+ written_ids.add(target_ann.image_id)
273
+
241
274
  write_result = target_handler.write_one(target_ann, write_dir)
242
275
  if not write_result.success:
243
276
  err = f"Failed to write {target_ann.image_id}: {write_result.message}"
@@ -220,6 +220,16 @@ class CocoAndLabelMeConverter(BaseConverter):
220
220
  Only the structural representation differs — coordinate values pass
221
221
  through unchanged.
222
222
  """
223
+ # COCO → LabelMe only: rewrite image_id to the COCO file_name stem so
224
+ # the output .json shares the stem with the image file (LabelMe format
225
+ # requirement; leading zeros preserved). LabelMe → COCO must NOT
226
+ # rewrite — the COCO output id derives from the numeric image_id.
227
+ new_image_id = image_ann.image_id
228
+ if self.source_format == "coco":
229
+ from .utils import coco_file_name_to_image_id
230
+
231
+ new_image_id = coco_file_name_to_image_id(image_ann.image_path, image_ann.image_id)
232
+
223
233
  new_objects = []
224
234
  for obj in image_ann.objects:
225
235
  new_bbox = None
@@ -251,7 +261,7 @@ class CocoAndLabelMeConverter(BaseConverter):
251
261
  )
252
262
 
253
263
  return ImageAnnotation(
254
- image_id=image_ann.image_id,
264
+ image_id=new_image_id,
255
265
  image_path=image_ann.image_path,
256
266
  width=image_ann.width,
257
267
  height=image_ann.height,
@@ -65,9 +65,62 @@ def format_convert_phase(phase: str, stats: Dict[str, Any]) -> str:
65
65
  return "\n".join(lines)
66
66
 
67
67
 
68
+ def format_output_layout(target_path: str) -> str:
69
+ """Return a compact summary of the generated output layout.
70
+
71
+ Scans the actual target path and lists its top-level entries:
72
+ directories show their recursive file count (``labels/ 4 files``);
73
+ empty directories show ``empty`` — an empty ``images/`` additionally
74
+ notes that images are not copied; single-file targets (COCO JSON)
75
+ show the file name.
76
+
77
+ Args:
78
+ target_path: Target output path of a conversion.
79
+
80
+ Returns:
81
+ ``"Output layout:"`` block string, or ``""`` when the target
82
+ path does not exist.
83
+ """
84
+ from pathlib import Path
85
+
86
+ target = Path(target_path)
87
+ if not target.exists():
88
+ return ""
89
+
90
+ lines = ["Output layout:"]
91
+ if target.is_file():
92
+ lines.append(f" {target.name} (single file)")
93
+ return "\n".join(lines)
94
+
95
+ try:
96
+ entries = sorted(target.iterdir(), key=lambda p: (p.is_file(), p.name.lower()))
97
+ except OSError:
98
+ return ""
99
+
100
+ for entry in entries:
101
+ if entry.is_dir():
102
+ count = sum(1 for p in entry.rglob("*") if p.is_file())
103
+ if count == 0:
104
+ if entry.name == "images":
105
+ lines.append(
106
+ f" {entry.name}/ empty — images are not copied; "
107
+ "place your image files here"
108
+ )
109
+ else:
110
+ lines.append(f" {entry.name}/ empty")
111
+ else:
112
+ lines.append(f" {entry.name}/ {count} files")
113
+ else:
114
+ lines.append(f" {entry.name}")
115
+ return "\n".join(lines)
116
+
117
+
68
118
  def format_convert_result(result: Any) -> str:
69
119
  """Return a final result block for a completed conversion.
70
120
 
121
+ The block is followed by an output layout summary (see
122
+ ``format_output_layout()``) when the target path exists.
123
+
71
124
  Args:
72
125
  result: A ``ConversionResult`` instance.
73
126
 
@@ -90,4 +143,11 @@ def format_convert_result(result: Any) -> str:
90
143
  if result.warnings:
91
144
  items["Warnings"] = len(result.warnings)
92
145
 
93
- return format_result_block(status, items, log_path=result.log_path)
146
+ block = format_result_block(status, items, log_path=result.log_path)
147
+
148
+ if getattr(result, "target_path", None):
149
+ layout = format_output_layout(result.target_path)
150
+ if layout:
151
+ block = f"{block}\n\n{layout}"
152
+
153
+ return block
@@ -130,6 +130,42 @@ def absolute_pixel_to_yolo(
130
130
  # ---------------------------------------------------------------------------
131
131
 
132
132
 
133
+ def coco_file_name_to_image_id(file_name: Optional[str], fallback_id: str) -> str:
134
+ """Derive the per-file output ``image_id`` from a COCO ``file_name``.
135
+
136
+ COCO ``file_name`` is the authoritative source of the image file stem
137
+ (``id`` is only a numeric reference key and carries no filename
138
+ information). Per-file output formats (YOLO `.txt`, LabelMe `.json`)
139
+ must share the stem with the image file (see `spec_yolo_format.md` /
140
+ `spec_labelme_format.md`), so COCO-source converters rewrite ``image_id``
141
+ to this stem before ``write_one()``.
142
+
143
+ Rules:
144
+ - Backslashes are normalized to ``/`` (Windows-style paths)
145
+ - Subdirectory components are dropped (output is flat; the
146
+ `spec_label.md` image_id invariant forbids path separators)
147
+ - Leading zeros are preserved (``000001.jpg`` → ``000001``)
148
+ - Degenerate stems (``""``, ``"."``, ``".."``) fall back to *fallback_id*
149
+ (the numeric COCO id string)
150
+
151
+ Args:
152
+ file_name: COCO ``images[].file_name`` value. ``None`` is tolerated
153
+ defensively and treated as empty (falls back to *fallback_id*).
154
+ fallback_id: ``ImageAnnotation.image_id`` from the COCO handler
155
+ (numeric id string), used when the stem is degenerate.
156
+
157
+ Returns:
158
+ Basename stem of *file_name*, or *fallback_id*.
159
+ """
160
+ if not file_name:
161
+ return fallback_id
162
+ normalized = str(file_name).replace("\\", "/")
163
+ stem = Path(normalized).stem
164
+ if stem in ("", ".", ".."):
165
+ return fallback_id
166
+ return stem
167
+
168
+
133
169
  def ensure_coco_categories_for_streaming(
134
170
  converter: Any,
135
171
  source_handler: Any,
@@ -288,7 +288,7 @@ class YoloAndCocoConverter(BaseConverter):
288
288
 
289
289
  def _coco_to_yolo_one(self, img: ImageAnnotation) -> ImageAnnotation:
290
290
  """Convert single image: COCO absolute px → YOLO normalized center."""
291
- from .utils import absolute_pixel_to_yolo
291
+ from .utils import absolute_pixel_to_yolo, coco_file_name_to_image_id
292
292
 
293
293
  new_objects = []
294
294
  for obj in img.objects:
@@ -306,8 +306,11 @@ class YoloAndCocoConverter(BaseConverter):
306
306
  )
307
307
  )
308
308
 
309
+ # YOLO format requires label files to share the stem with the image
310
+ # file — derive the output name from the COCO file_name stem (which
311
+ # preserves leading zeros), not from the numeric COCO id.
309
312
  return ImageAnnotation(
310
- image_id=img.image_id,
313
+ image_id=coco_file_name_to_image_id(img.image_path, img.image_id),
311
314
  image_path=img.image_path,
312
315
  width=img.width,
313
316
  height=img.height,
@@ -203,11 +203,17 @@ def _dataset_to_coco_dict(dataset: Any) -> Dict[str, Any]:
203
203
  # Preserve top-level keys from dataset_info
204
204
  info = dataset.dataset_info.copy()
205
205
 
206
- # Reconstruct images
206
+ # Reconstruct images — image ids resolve with the same rule as the COCO
207
+ # handler write() (digit image_id → int, non-digit/colliding → dedicated
208
+ # counter, unique ids guaranteed — resolve_coco_image_ids()).
209
+ from dataflow.label.utils import resolve_coco_image_ids
210
+
211
+ id_map = resolve_coco_image_ids(img.image_id for img in dataset.images)
212
+
207
213
  images = []
208
214
  for img in dataset.images:
209
215
  img_entry = {
210
- "id": int(img.image_id) if img.image_id.isdigit() else img.image_id,
216
+ "id": id_map[img.image_id],
211
217
  "file_name": img.image_path,
212
218
  "width": img.width,
213
219
  "height": img.height,
@@ -223,7 +229,7 @@ def _dataset_to_coco_dict(dataset: Any) -> Dict[str, Any]:
223
229
  annotations = []
224
230
  ann_id = 1
225
231
  for img in dataset.images:
226
- image_id = int(img.image_id) if str(img.image_id).isdigit() else img.image_id
232
+ image_id = id_map[img.image_id]
227
233
  for obj in img.objects:
228
234
  ann_entry: Dict[str, Any] = {
229
235
  "id": ann_id,
@@ -607,12 +607,18 @@ class CocoAnnotationHandler(BaseAnnotationHandler):
607
607
  images = []
608
608
  coco_annotations = []
609
609
  ann_id = 1
610
- img_id_counter = 1
610
+
611
+ # Resolve format-native image_id strings to unique positive-int COCO
612
+ # ids (digit strings keep their value; non-digit and colliding ids
613
+ # get dedicated counter ids — see resolve_coco_image_ids()).
614
+ from .utils import resolve_coco_image_ids
615
+
616
+ id_map = resolve_coco_image_ids(img.image_id for img in annotations.images)
611
617
 
612
618
  for img in annotations.images:
613
619
  images.append(
614
620
  {
615
- "id": int(img.image_id) if img.image_id.isdigit() else img_id_counter,
621
+ "id": id_map[img.image_id],
616
622
  "width": img.width,
617
623
  "height": img.height,
618
624
  "file_name": img.image_path,
@@ -625,7 +631,7 @@ class CocoAnnotationHandler(BaseAnnotationHandler):
625
631
 
626
632
  # Add object annotations
627
633
  for obj in img.objects:
628
- coco_ann = self._object_to_coco_annotation(obj, img, ann_id, img_id_counter)
634
+ coco_ann = self._object_to_coco_annotation(obj, img, ann_id, id_map[img.image_id])
629
635
  if coco_ann:
630
636
  coco_annotations.append(coco_ann)
631
637
  ann_id += 1
@@ -639,8 +645,6 @@ class CocoAnnotationHandler(BaseAnnotationHandler):
639
645
  f"Skipping object {obj.class_name}: conversion to COCO format failed"
640
646
  )
641
647
 
642
- img_id_counter += 1
643
-
644
648
  # Prediction mode: output plain list of annotation dicts
645
649
  if self.prediction:
646
650
  return coco_annotations
@@ -661,9 +665,14 @@ class CocoAnnotationHandler(BaseAnnotationHandler):
661
665
  return result
662
666
 
663
667
  def _object_to_coco_annotation(
664
- self, obj: ObjectAnnotation, img: ImageAnnotation, ann_id: int, img_id: int
668
+ self, obj: ObjectAnnotation, img: ImageAnnotation, ann_id: int, resolved_img_id: int
665
669
  ) -> Optional[Dict]:
666
- """Convert ObjectAnnotation to COCO annotation dict."""
670
+ """Convert ObjectAnnotation to COCO annotation dict.
671
+
672
+ Args:
673
+ resolved_img_id: COCO image id resolved from ``img.image_id`` by
674
+ ``resolve_coco_image_ids()`` (see ``_prepare_coco_data``).
675
+ """
667
676
  try:
668
677
  # Determine segmentation format
669
678
  segmentation = None
@@ -767,10 +776,8 @@ class CocoAnnotationHandler(BaseAnnotationHandler):
767
776
  bbox = [float(min_x), float(min_y), float(w), float(h)]
768
777
  area = float(w * h)
769
778
 
770
- image_id_val = int(img.image_id) if img.image_id.isdigit() else img_id
771
-
772
779
  ann_dict = {
773
- "image_id": image_id_val,
780
+ "image_id": resolved_img_id,
774
781
  "category_id": obj.class_id,
775
782
  "segmentation": segmentation,
776
783
  "area": area,
@@ -4,7 +4,52 @@ Utility functions for the label module.
4
4
 
5
5
  import hashlib
6
6
  from pathlib import Path
7
- from typing import Dict, Optional
7
+ from typing import Dict, Iterable, Optional, Set
8
+
9
+
10
+ def resolve_coco_image_ids(image_ids: Iterable[str]) -> Dict[str, int]:
11
+ """Resolve format-native ``image_id`` strings to unique positive-int COCO ids.
12
+
13
+ COCO ``images[].id`` must be an int and unique per image
14
+ (`spec_coco_format.md`). ``ImageAnnotation.image_id`` is a string and may
15
+ be a file stem, so COCO writers resolve it:
16
+
17
+ - Digit-string ``image_id`` → its ``int`` value (preserves roundtrip ids)
18
+ - Non-digit ``image_id`` → a dedicated counter id (not the annotation
19
+ counter); counters skip values reserved by digit-string ids
20
+ - Uniqueness guaranteed: if a resolved id collides (e.g. ``"01"`` vs
21
+ ``"1"``), the later image gets a fresh counter id
22
+
23
+ Canonical implementation — used by both
24
+ ``CocoAnnotationHandler._prepare_coco_data()`` and
25
+ ``evaluate/utils._dataset_to_coco_dict()``.
26
+
27
+ Args:
28
+ image_ids: Iterable of image_id strings in output order.
29
+
30
+ Returns:
31
+ Mapping ``{image_id_string: resolved_int_id}``.
32
+ """
33
+ ids = [str(s) for s in image_ids]
34
+ # Values reserved by digit-string ids — counters must never take them,
35
+ # otherwise a later digit id would collide with an earlier counter.
36
+ digit_values: Set[int] = {int(s) for s in ids if s.isdigit()}
37
+ used: Set[int] = set()
38
+ mapping: Dict[str, int] = {}
39
+ counter = 1
40
+
41
+ for s in ids:
42
+ if s.isdigit() and int(s) not in used:
43
+ value = int(s)
44
+ else:
45
+ while counter in used or counter in digit_values:
46
+ counter += 1
47
+ value = counter
48
+ counter += 1
49
+ used.add(value)
50
+ mapping[s] = value
51
+
52
+ return mapping
8
53
 
9
54
 
10
55
  def parse_yolo_class_id(token: str) -> Optional[int]:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dataflow-cv
3
- Version: 2.0.1
3
+ Version: 3.0.0
4
4
  Summary: A computer vision dataset processing library — analyse, convert, visualize, and evaluate annotations across YOLO, LabelMe, and COCO formats
5
5
  Author: DataFlow-CV Team
6
6
  License: MIT
@@ -156,6 +156,12 @@ dataflow-cv convert yolo2coco --verbose images/ labels/ classes.txt output.json
156
156
  dataflow-cv convert yolo2coco --no-strict images/ labels/ classes.txt output.json
157
157
  ```
158
158
 
159
+ > 📂 **COCO→YOLO / COCO→LabelMe output naming**: each `.txt` / `.json` is named by the
160
+ > image file stem — leading zeros preserved (`000001.jpg` → `000001.txt`). The
161
+ > converter generates `labels/` (or the `.json` files) + `classes.txt`; the
162
+ > `images/` directory is created but left **empty** — images are never copied,
163
+ > place your image files there yourself.
164
+
159
165
  #### 🎨 Visualization
160
166
 
161
167
  ```bash
@@ -231,7 +237,13 @@ Two evaluation modes, distinguished by how overlap is measured:
231
237
 
232
238
  ```python
233
239
  from dataflow.util.logging import LogConfig
234
- from dataflow.analyse import StatsAnalyser, SplitAnalyser, FilterAnalyser, PartitionAnalyser, SampleAnalyser
240
+ from dataflow.analyse import (
241
+ StatsAnalyser,
242
+ SplitAnalyser,
243
+ FilterAnalyser,
244
+ PartitionAnalyser,
245
+ SampleAnalyser,
246
+ )
235
247
  from dataflow.convert import YoloAndCocoConverter
236
248
  from dataflow.visualize import YOLOVisualizer
237
249
  from dataflow.evaluate import DetectionEvaluator, compute_pr_f1
@@ -247,37 +259,50 @@ print(f"{result.data.total_files} images, {result.data.total_annotations} object
247
259
  # Train/test split (YOLO / LabelMe)
248
260
  splitter = SplitAnalyser(log_config=log_cfg)
249
261
  result = splitter.analyse(
250
- output_dir="output/", ratio=0.8, seed=42,
251
- label_dir="yolo_labels/", class_file="classes.txt",
262
+ output_dir="output/",
263
+ ratio=0.8,
264
+ seed=42,
265
+ label_dir="yolo_labels/",
266
+ class_file="classes.txt",
252
267
  )
253
268
  print(f"Train: {result.data.train_count}, Val: {result.data.val_count}")
254
269
 
255
270
  # Split with images (both mode — labels drive, images follow by stem)
256
271
  result = splitter.analyse(
257
- output_dir="output/", ratio=0.8, seed=42,
258
- label_dir="yolo_labels/", image_dir="images/",
272
+ output_dir="output/",
273
+ ratio=0.8,
274
+ seed=42,
275
+ label_dir="yolo_labels/",
276
+ image_dir="images/",
259
277
  class_file="classes.txt",
260
278
  )
261
279
 
262
280
  # Category filter (keep / remap categories per new classes.txt)
263
281
  filterer = FilterAnalyser(log_config=log_cfg)
264
282
  result = filterer.analyse(
265
- "yolo_labels/", original_class_file="classes.txt",
266
- new_class_file="classes_new.txt", output_dir="filtered/",
283
+ "yolo_labels/",
284
+ original_class_file="classes.txt",
285
+ new_class_file="classes_new.txt",
286
+ output_dir="filtered/",
267
287
  )
268
288
 
269
289
  # N-way partition (YOLO / LabelMe labels; images follow by stem)
270
290
  partitioner = PartitionAnalyser(log_config=log_cfg)
271
291
  result = partitioner.analyse(
272
- output_dir="parts/", num=4,
273
- label_dir="yolo_labels/", image_dir="images/",
292
+ output_dir="parts/",
293
+ num=4,
294
+ label_dir="yolo_labels/",
295
+ image_dir="images/",
274
296
  )
275
297
 
276
298
  # File sampling (labels, images, or both — random or sequential)
277
299
  sampler = SampleAnalyser(log_config=log_cfg)
278
300
  result = sampler.analyse(
279
- output_dir="sampled/", count=10,
280
- label_dir="yolo_labels/", shuffle=True, seed=42,
301
+ output_dir="sampled/",
302
+ count=10,
303
+ label_dir="yolo_labels/",
304
+ shuffle=True,
305
+ seed=42,
281
306
  )
282
307
 
283
308
  # ── Convert ──────────────────────────────────────────
@@ -285,22 +310,30 @@ result = sampler.analyse(
285
310
  log_cfg = LogConfig(name="convert", verbose=True)
286
311
  converter = YoloAndCocoConverter(source_to_target=True, log_config=log_cfg, strict_mode=True)
287
312
  result = converter.convert(
288
- source_path="yolo_labels/", target_path="anno.json",
289
- class_file="classes.txt", image_dir="images/",
313
+ source_path="yolo_labels/",
314
+ target_path="anno.json",
315
+ class_file="classes.txt",
316
+ image_dir="images/",
290
317
  )
291
318
 
292
319
  # YOLO predictions → COCO (prediction mode)
293
320
  converter = YoloAndCocoConverter(source_to_target=True, prediction=True)
294
321
  result = converter.convert(
295
- source_path="yolo_preds/", target_path="pred.json",
296
- class_file="classes.txt", image_dir="images/",
322
+ source_path="yolo_preds/",
323
+ target_path="pred.json",
324
+ class_file="classes.txt",
325
+ image_dir="images/",
297
326
  )
298
327
 
299
328
  # ── Visualize ────────────────────────────────────────
300
329
  visualizer = YOLOVisualizer(
301
- label_dir="yolo_labels/", image_dir="images/",
302
- class_file="classes.txt", is_show=True, is_save=True,
303
- output_dir="visualized/", log_config=log_cfg,
330
+ label_dir="yolo_labels/",
331
+ image_dir="images/",
332
+ class_file="classes.txt",
333
+ is_show=True,
334
+ is_save=True,
335
+ output_dir="visualized/",
336
+ log_config=log_cfg,
304
337
  )
305
338
  result = visualizer.visualize()
306
339
 
@@ -370,7 +403,7 @@ For detailed developer guidance including advanced test commands, debugging, and
370
403
 
371
404
  ### 🧪 Testing
372
405
 
373
- **561 tests, 80% code coverage (5462 statements).**
406
+ **606 tests, 80% code coverage (5532 statements).**
374
407
 
375
408
  ```bash
376
409
  pytest # All tests
@@ -384,11 +417,11 @@ pytest tests/evaluate/test_evaluator.py # Single module
384
417
 
385
418
  | Module | Coverage | Highlights |
386
419
  |--------|:--------:|------------|
387
- | `dataflow/label/` | 71% | models (84%), base (82%), utils (78%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
388
- | `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (85%), split (85%), stats (83%), filter (76%), partition (74%) |
389
- | `dataflow/convert/` | 85% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (87%), coco_and_labelme (86%), base (80%), rle (80%) |
420
+ | `dataflow/label/` | 71% | models (84%), base (82%), utils (83%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
421
+ | `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (84%), split (85%), stats (82%), filter (76%), partition (74%) |
422
+ | `dataflow/convert/` | 86% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (88%), coco_and_labelme (88%), log_templates (85%), base (81%), rle (80%) |
390
423
  | `dataflow/visualize/` | 81% | yolo_vis (100%), labelme_vis (100%), coco_vis (93%), base (76%) |
391
- | `dataflow/evaluate/` | 87% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (67%) |
424
+ | `dataflow/evaluate/` | 90% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (87%) |
392
425
  | `dataflow/cli/` | 74% | main (96%), visualize cmd (87%), utils (87%), evaluate cmd (83%), analyse cmd (65%), convert cmd (52%) |
393
426
  | `dataflow/util/` | 100% | logging (100%) |
394
427
 
@@ -397,11 +430,10 @@ pytest tests/evaluate/test_evaluator.py # Single module
397
430
  ### 🎨 Code Quality
398
431
 
399
432
  ```bash
400
- pip install -e .[dev] # Install dev dependencies
401
- black dataflow tests samples # Format
402
- isort dataflow tests samples # Sort imports
403
- mypy dataflow # Type check
404
- flake8 dataflow tests samples # Lint
433
+ pip install -e .[dev] # Install dev dependencies
434
+ ruff check dataflow tests samples # Lint
435
+ ruff format --check dataflow tests samples # Format check
436
+ mypy dataflow # Type check
405
437
  ```
406
438
 
407
439
  ### 🔗 Pre-commit Hooks (Optional)
@@ -411,7 +443,7 @@ pip install pre-commit
411
443
  pre-commit install # Install git hooks (run once)
412
444
 
413
445
  # After this, every `git commit` auto-runs:
414
- # black → isort → flake8 → whitespace checks
446
+ # ruff (lint, auto-fix) → ruff format → whitespace checks
415
447
 
416
448
  pre-commit run --all-files # Manual run against all files
417
449
  ```
@@ -427,7 +459,7 @@ dataflow/
427
459
  ├── evaluate/ # pycocotools-based metrics, log templates
428
460
  ├── util/ # Unified logging (LogManager + format helpers)
429
461
  └── cli/ # CLI entry point, commands, validation
430
- tests/ # Unit & integration tests (561 tests, conftest fixtures)
462
+ tests/ # Unit & integration tests (606 tests, conftest fixtures)
431
463
  samples/ # Python API usage examples (analyse, convert, visualize, evaluate, cli)
432
464
  assets/ # Test data (det/seg by format)
433
465
  specs/ # Canonical specifications (evaluate/ + formats/ + modules/)
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "dataflow-cv"
7
- version = "2.0.1"
7
+ version = "3.0.0"
8
8
  description = "A computer vision dataset processing library — analyse, convert, visualize, and evaluate annotations across YOLO, LabelMe, and COCO formats"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.8"
File without changes
File without changes