dataflow-cv 2.0.1__tar.gz → 3.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataflow_cv-2.0.1/dataflow_cv.egg-info → dataflow_cv-3.0.0}/PKG-INFO +63 -31
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/README.md +62 -30
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/__init__.py +1 -1
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/base.py +34 -1
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/coco_and_labelme.py +11 -1
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/log_templates.py +61 -1
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/utils.py +36 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/yolo_and_coco.py +5 -2
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/utils.py +9 -3
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/coco_handler.py +17 -10
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/utils.py +46 -1
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0/dataflow_cv.egg-info}/PKG-INFO +63 -31
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/pyproject.toml +1 -1
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/LICENSE +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/__init__.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/base.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/filter.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/log_templates.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/partition.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/sample.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/split.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/stats.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/analyse/utils.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/__init__.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/__init__.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/analyse.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/convert.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/evaluate.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/utils.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/commands/visualize.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/exceptions.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/cli/main.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/__init__.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/labelme_and_yolo.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/convert/rle_converter.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/__init__.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/base.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/evaluator.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/log_templates.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/metrics.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/evaluate/result.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/__init__.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/base.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/labelme_handler.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/models.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/label/yolo_handler.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/util/__init__.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/util/logging.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/__init__.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/base.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/coco_visualizer.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/labelme_visualizer.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/log_templates.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/utils.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow/visualize/yolo_visualizer.py +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/SOURCES.txt +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/dependency_links.txt +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/entry_points.txt +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/not-zip-safe +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/requires.txt +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/top_level.txt +0 -0
- {dataflow_cv-2.0.1 → dataflow_cv-3.0.0}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dataflow-cv
|
|
3
|
-
Version:
|
|
3
|
+
Version: 3.0.0
|
|
4
4
|
Summary: A computer vision dataset processing library — analyse, convert, visualize, and evaluate annotations across YOLO, LabelMe, and COCO formats
|
|
5
5
|
Author: DataFlow-CV Team
|
|
6
6
|
License: MIT
|
|
@@ -156,6 +156,12 @@ dataflow-cv convert yolo2coco --verbose images/ labels/ classes.txt output.json
|
|
|
156
156
|
dataflow-cv convert yolo2coco --no-strict images/ labels/ classes.txt output.json
|
|
157
157
|
```
|
|
158
158
|
|
|
159
|
+
> 📂 **COCO→YOLO / COCO→LabelMe output naming**: each `.txt` / `.json` is named by the
|
|
160
|
+
> image file stem — leading zeros preserved (`000001.jpg` → `000001.txt`). The
|
|
161
|
+
> converter generates `labels/` (or the `.json` files) + `classes.txt`; the
|
|
162
|
+
> `images/` directory is created but left **empty** — images are never copied,
|
|
163
|
+
> place your image files there yourself.
|
|
164
|
+
|
|
159
165
|
#### 🎨 Visualization
|
|
160
166
|
|
|
161
167
|
```bash
|
|
@@ -231,7 +237,13 @@ Two evaluation modes, distinguished by how overlap is measured:
|
|
|
231
237
|
|
|
232
238
|
```python
|
|
233
239
|
from dataflow.util.logging import LogConfig
|
|
234
|
-
from dataflow.analyse import
|
|
240
|
+
from dataflow.analyse import (
|
|
241
|
+
StatsAnalyser,
|
|
242
|
+
SplitAnalyser,
|
|
243
|
+
FilterAnalyser,
|
|
244
|
+
PartitionAnalyser,
|
|
245
|
+
SampleAnalyser,
|
|
246
|
+
)
|
|
235
247
|
from dataflow.convert import YoloAndCocoConverter
|
|
236
248
|
from dataflow.visualize import YOLOVisualizer
|
|
237
249
|
from dataflow.evaluate import DetectionEvaluator, compute_pr_f1
|
|
@@ -247,37 +259,50 @@ print(f"{result.data.total_files} images, {result.data.total_annotations} object
|
|
|
247
259
|
# Train/test split (YOLO / LabelMe)
|
|
248
260
|
splitter = SplitAnalyser(log_config=log_cfg)
|
|
249
261
|
result = splitter.analyse(
|
|
250
|
-
output_dir="output/",
|
|
251
|
-
|
|
262
|
+
output_dir="output/",
|
|
263
|
+
ratio=0.8,
|
|
264
|
+
seed=42,
|
|
265
|
+
label_dir="yolo_labels/",
|
|
266
|
+
class_file="classes.txt",
|
|
252
267
|
)
|
|
253
268
|
print(f"Train: {result.data.train_count}, Val: {result.data.val_count}")
|
|
254
269
|
|
|
255
270
|
# Split with images (both mode — labels drive, images follow by stem)
|
|
256
271
|
result = splitter.analyse(
|
|
257
|
-
output_dir="output/",
|
|
258
|
-
|
|
272
|
+
output_dir="output/",
|
|
273
|
+
ratio=0.8,
|
|
274
|
+
seed=42,
|
|
275
|
+
label_dir="yolo_labels/",
|
|
276
|
+
image_dir="images/",
|
|
259
277
|
class_file="classes.txt",
|
|
260
278
|
)
|
|
261
279
|
|
|
262
280
|
# Category filter (keep / remap categories per new classes.txt)
|
|
263
281
|
filterer = FilterAnalyser(log_config=log_cfg)
|
|
264
282
|
result = filterer.analyse(
|
|
265
|
-
"yolo_labels/",
|
|
266
|
-
|
|
283
|
+
"yolo_labels/",
|
|
284
|
+
original_class_file="classes.txt",
|
|
285
|
+
new_class_file="classes_new.txt",
|
|
286
|
+
output_dir="filtered/",
|
|
267
287
|
)
|
|
268
288
|
|
|
269
289
|
# N-way partition (YOLO / LabelMe labels; images follow by stem)
|
|
270
290
|
partitioner = PartitionAnalyser(log_config=log_cfg)
|
|
271
291
|
result = partitioner.analyse(
|
|
272
|
-
output_dir="parts/",
|
|
273
|
-
|
|
292
|
+
output_dir="parts/",
|
|
293
|
+
num=4,
|
|
294
|
+
label_dir="yolo_labels/",
|
|
295
|
+
image_dir="images/",
|
|
274
296
|
)
|
|
275
297
|
|
|
276
298
|
# File sampling (labels, images, or both — random or sequential)
|
|
277
299
|
sampler = SampleAnalyser(log_config=log_cfg)
|
|
278
300
|
result = sampler.analyse(
|
|
279
|
-
output_dir="sampled/",
|
|
280
|
-
|
|
301
|
+
output_dir="sampled/",
|
|
302
|
+
count=10,
|
|
303
|
+
label_dir="yolo_labels/",
|
|
304
|
+
shuffle=True,
|
|
305
|
+
seed=42,
|
|
281
306
|
)
|
|
282
307
|
|
|
283
308
|
# ── Convert ──────────────────────────────────────────
|
|
@@ -285,22 +310,30 @@ result = sampler.analyse(
|
|
|
285
310
|
log_cfg = LogConfig(name="convert", verbose=True)
|
|
286
311
|
converter = YoloAndCocoConverter(source_to_target=True, log_config=log_cfg, strict_mode=True)
|
|
287
312
|
result = converter.convert(
|
|
288
|
-
source_path="yolo_labels/",
|
|
289
|
-
|
|
313
|
+
source_path="yolo_labels/",
|
|
314
|
+
target_path="anno.json",
|
|
315
|
+
class_file="classes.txt",
|
|
316
|
+
image_dir="images/",
|
|
290
317
|
)
|
|
291
318
|
|
|
292
319
|
# YOLO predictions → COCO (prediction mode)
|
|
293
320
|
converter = YoloAndCocoConverter(source_to_target=True, prediction=True)
|
|
294
321
|
result = converter.convert(
|
|
295
|
-
source_path="yolo_preds/",
|
|
296
|
-
|
|
322
|
+
source_path="yolo_preds/",
|
|
323
|
+
target_path="pred.json",
|
|
324
|
+
class_file="classes.txt",
|
|
325
|
+
image_dir="images/",
|
|
297
326
|
)
|
|
298
327
|
|
|
299
328
|
# ── Visualize ────────────────────────────────────────
|
|
300
329
|
visualizer = YOLOVisualizer(
|
|
301
|
-
label_dir="yolo_labels/",
|
|
302
|
-
|
|
303
|
-
|
|
330
|
+
label_dir="yolo_labels/",
|
|
331
|
+
image_dir="images/",
|
|
332
|
+
class_file="classes.txt",
|
|
333
|
+
is_show=True,
|
|
334
|
+
is_save=True,
|
|
335
|
+
output_dir="visualized/",
|
|
336
|
+
log_config=log_cfg,
|
|
304
337
|
)
|
|
305
338
|
result = visualizer.visualize()
|
|
306
339
|
|
|
@@ -370,7 +403,7 @@ For detailed developer guidance including advanced test commands, debugging, and
|
|
|
370
403
|
|
|
371
404
|
### 🧪 Testing
|
|
372
405
|
|
|
373
|
-
**
|
|
406
|
+
**606 tests, 80% code coverage (5532 statements).**
|
|
374
407
|
|
|
375
408
|
```bash
|
|
376
409
|
pytest # All tests
|
|
@@ -384,11 +417,11 @@ pytest tests/evaluate/test_evaluator.py # Single module
|
|
|
384
417
|
|
|
385
418
|
| Module | Coverage | Highlights |
|
|
386
419
|
|--------|:--------:|------------|
|
|
387
|
-
| `dataflow/label/` | 71% | models (84%), base (82%), utils (
|
|
388
|
-
| `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (
|
|
389
|
-
| `dataflow/convert/` |
|
|
420
|
+
| `dataflow/label/` | 71% | models (84%), base (82%), utils (83%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
|
|
421
|
+
| `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (84%), split (85%), stats (82%), filter (76%), partition (74%) |
|
|
422
|
+
| `dataflow/convert/` | 86% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (88%), coco_and_labelme (88%), log_templates (85%), base (81%), rle (80%) |
|
|
390
423
|
| `dataflow/visualize/` | 81% | yolo_vis (100%), labelme_vis (100%), coco_vis (93%), base (76%) |
|
|
391
|
-
| `dataflow/evaluate/` |
|
|
424
|
+
| `dataflow/evaluate/` | 90% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (87%) |
|
|
392
425
|
| `dataflow/cli/` | 74% | main (96%), visualize cmd (87%), utils (87%), evaluate cmd (83%), analyse cmd (65%), convert cmd (52%) |
|
|
393
426
|
| `dataflow/util/` | 100% | logging (100%) |
|
|
394
427
|
|
|
@@ -397,11 +430,10 @@ pytest tests/evaluate/test_evaluator.py # Single module
|
|
|
397
430
|
### 🎨 Code Quality
|
|
398
431
|
|
|
399
432
|
```bash
|
|
400
|
-
pip install -e .[dev]
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
mypy dataflow
|
|
404
|
-
flake8 dataflow tests samples # Lint
|
|
433
|
+
pip install -e .[dev] # Install dev dependencies
|
|
434
|
+
ruff check dataflow tests samples # Lint
|
|
435
|
+
ruff format --check dataflow tests samples # Format check
|
|
436
|
+
mypy dataflow # Type check
|
|
405
437
|
```
|
|
406
438
|
|
|
407
439
|
### 🔗 Pre-commit Hooks (Optional)
|
|
@@ -411,7 +443,7 @@ pip install pre-commit
|
|
|
411
443
|
pre-commit install # Install git hooks (run once)
|
|
412
444
|
|
|
413
445
|
# After this, every `git commit` auto-runs:
|
|
414
|
-
#
|
|
446
|
+
# ruff (lint, auto-fix) → ruff format → whitespace checks
|
|
415
447
|
|
|
416
448
|
pre-commit run --all-files # Manual run against all files
|
|
417
449
|
```
|
|
@@ -427,7 +459,7 @@ dataflow/
|
|
|
427
459
|
├── evaluate/ # pycocotools-based metrics, log templates
|
|
428
460
|
├── util/ # Unified logging (LogManager + format helpers)
|
|
429
461
|
└── cli/ # CLI entry point, commands, validation
|
|
430
|
-
tests/ # Unit & integration tests (
|
|
462
|
+
tests/ # Unit & integration tests (606 tests, conftest fixtures)
|
|
431
463
|
samples/ # Python API usage examples (analyse, convert, visualize, evaluate, cli)
|
|
432
464
|
assets/ # Test data (det/seg by format)
|
|
433
465
|
specs/ # Canonical specifications (evaluate/ + formats/ + modules/)
|
|
@@ -117,6 +117,12 @@ dataflow-cv convert yolo2coco --verbose images/ labels/ classes.txt output.json
|
|
|
117
117
|
dataflow-cv convert yolo2coco --no-strict images/ labels/ classes.txt output.json
|
|
118
118
|
```
|
|
119
119
|
|
|
120
|
+
> 📂 **COCO→YOLO / COCO→LabelMe output naming**: each `.txt` / `.json` is named by the
|
|
121
|
+
> image file stem — leading zeros preserved (`000001.jpg` → `000001.txt`). The
|
|
122
|
+
> converter generates `labels/` (or the `.json` files) + `classes.txt`; the
|
|
123
|
+
> `images/` directory is created but left **empty** — images are never copied,
|
|
124
|
+
> place your image files there yourself.
|
|
125
|
+
|
|
120
126
|
#### 🎨 Visualization
|
|
121
127
|
|
|
122
128
|
```bash
|
|
@@ -192,7 +198,13 @@ Two evaluation modes, distinguished by how overlap is measured:
|
|
|
192
198
|
|
|
193
199
|
```python
|
|
194
200
|
from dataflow.util.logging import LogConfig
|
|
195
|
-
from dataflow.analyse import
|
|
201
|
+
from dataflow.analyse import (
|
|
202
|
+
StatsAnalyser,
|
|
203
|
+
SplitAnalyser,
|
|
204
|
+
FilterAnalyser,
|
|
205
|
+
PartitionAnalyser,
|
|
206
|
+
SampleAnalyser,
|
|
207
|
+
)
|
|
196
208
|
from dataflow.convert import YoloAndCocoConverter
|
|
197
209
|
from dataflow.visualize import YOLOVisualizer
|
|
198
210
|
from dataflow.evaluate import DetectionEvaluator, compute_pr_f1
|
|
@@ -208,37 +220,50 @@ print(f"{result.data.total_files} images, {result.data.total_annotations} object
|
|
|
208
220
|
# Train/test split (YOLO / LabelMe)
|
|
209
221
|
splitter = SplitAnalyser(log_config=log_cfg)
|
|
210
222
|
result = splitter.analyse(
|
|
211
|
-
output_dir="output/",
|
|
212
|
-
|
|
223
|
+
output_dir="output/",
|
|
224
|
+
ratio=0.8,
|
|
225
|
+
seed=42,
|
|
226
|
+
label_dir="yolo_labels/",
|
|
227
|
+
class_file="classes.txt",
|
|
213
228
|
)
|
|
214
229
|
print(f"Train: {result.data.train_count}, Val: {result.data.val_count}")
|
|
215
230
|
|
|
216
231
|
# Split with images (both mode — labels drive, images follow by stem)
|
|
217
232
|
result = splitter.analyse(
|
|
218
|
-
output_dir="output/",
|
|
219
|
-
|
|
233
|
+
output_dir="output/",
|
|
234
|
+
ratio=0.8,
|
|
235
|
+
seed=42,
|
|
236
|
+
label_dir="yolo_labels/",
|
|
237
|
+
image_dir="images/",
|
|
220
238
|
class_file="classes.txt",
|
|
221
239
|
)
|
|
222
240
|
|
|
223
241
|
# Category filter (keep / remap categories per new classes.txt)
|
|
224
242
|
filterer = FilterAnalyser(log_config=log_cfg)
|
|
225
243
|
result = filterer.analyse(
|
|
226
|
-
"yolo_labels/",
|
|
227
|
-
|
|
244
|
+
"yolo_labels/",
|
|
245
|
+
original_class_file="classes.txt",
|
|
246
|
+
new_class_file="classes_new.txt",
|
|
247
|
+
output_dir="filtered/",
|
|
228
248
|
)
|
|
229
249
|
|
|
230
250
|
# N-way partition (YOLO / LabelMe labels; images follow by stem)
|
|
231
251
|
partitioner = PartitionAnalyser(log_config=log_cfg)
|
|
232
252
|
result = partitioner.analyse(
|
|
233
|
-
output_dir="parts/",
|
|
234
|
-
|
|
253
|
+
output_dir="parts/",
|
|
254
|
+
num=4,
|
|
255
|
+
label_dir="yolo_labels/",
|
|
256
|
+
image_dir="images/",
|
|
235
257
|
)
|
|
236
258
|
|
|
237
259
|
# File sampling (labels, images, or both — random or sequential)
|
|
238
260
|
sampler = SampleAnalyser(log_config=log_cfg)
|
|
239
261
|
result = sampler.analyse(
|
|
240
|
-
output_dir="sampled/",
|
|
241
|
-
|
|
262
|
+
output_dir="sampled/",
|
|
263
|
+
count=10,
|
|
264
|
+
label_dir="yolo_labels/",
|
|
265
|
+
shuffle=True,
|
|
266
|
+
seed=42,
|
|
242
267
|
)
|
|
243
268
|
|
|
244
269
|
# ── Convert ──────────────────────────────────────────
|
|
@@ -246,22 +271,30 @@ result = sampler.analyse(
|
|
|
246
271
|
log_cfg = LogConfig(name="convert", verbose=True)
|
|
247
272
|
converter = YoloAndCocoConverter(source_to_target=True, log_config=log_cfg, strict_mode=True)
|
|
248
273
|
result = converter.convert(
|
|
249
|
-
source_path="yolo_labels/",
|
|
250
|
-
|
|
274
|
+
source_path="yolo_labels/",
|
|
275
|
+
target_path="anno.json",
|
|
276
|
+
class_file="classes.txt",
|
|
277
|
+
image_dir="images/",
|
|
251
278
|
)
|
|
252
279
|
|
|
253
280
|
# YOLO predictions → COCO (prediction mode)
|
|
254
281
|
converter = YoloAndCocoConverter(source_to_target=True, prediction=True)
|
|
255
282
|
result = converter.convert(
|
|
256
|
-
source_path="yolo_preds/",
|
|
257
|
-
|
|
283
|
+
source_path="yolo_preds/",
|
|
284
|
+
target_path="pred.json",
|
|
285
|
+
class_file="classes.txt",
|
|
286
|
+
image_dir="images/",
|
|
258
287
|
)
|
|
259
288
|
|
|
260
289
|
# ── Visualize ────────────────────────────────────────
|
|
261
290
|
visualizer = YOLOVisualizer(
|
|
262
|
-
label_dir="yolo_labels/",
|
|
263
|
-
|
|
264
|
-
|
|
291
|
+
label_dir="yolo_labels/",
|
|
292
|
+
image_dir="images/",
|
|
293
|
+
class_file="classes.txt",
|
|
294
|
+
is_show=True,
|
|
295
|
+
is_save=True,
|
|
296
|
+
output_dir="visualized/",
|
|
297
|
+
log_config=log_cfg,
|
|
265
298
|
)
|
|
266
299
|
result = visualizer.visualize()
|
|
267
300
|
|
|
@@ -331,7 +364,7 @@ For detailed developer guidance including advanced test commands, debugging, and
|
|
|
331
364
|
|
|
332
365
|
### 🧪 Testing
|
|
333
366
|
|
|
334
|
-
**
|
|
367
|
+
**606 tests, 80% code coverage (5532 statements).**
|
|
335
368
|
|
|
336
369
|
```bash
|
|
337
370
|
pytest # All tests
|
|
@@ -345,11 +378,11 @@ pytest tests/evaluate/test_evaluator.py # Single module
|
|
|
345
378
|
|
|
346
379
|
| Module | Coverage | Highlights |
|
|
347
380
|
|--------|:--------:|------------|
|
|
348
|
-
| `dataflow/label/` | 71% | models (84%), base (82%), utils (
|
|
349
|
-
| `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (
|
|
350
|
-
| `dataflow/convert/` |
|
|
381
|
+
| `dataflow/label/` | 71% | models (84%), base (82%), utils (83%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
|
|
382
|
+
| `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (84%), split (85%), stats (82%), filter (76%), partition (74%) |
|
|
383
|
+
| `dataflow/convert/` | 86% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (88%), coco_and_labelme (88%), log_templates (85%), base (81%), rle (80%) |
|
|
351
384
|
| `dataflow/visualize/` | 81% | yolo_vis (100%), labelme_vis (100%), coco_vis (93%), base (76%) |
|
|
352
|
-
| `dataflow/evaluate/` |
|
|
385
|
+
| `dataflow/evaluate/` | 90% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (87%) |
|
|
353
386
|
| `dataflow/cli/` | 74% | main (96%), visualize cmd (87%), utils (87%), evaluate cmd (83%), analyse cmd (65%), convert cmd (52%) |
|
|
354
387
|
| `dataflow/util/` | 100% | logging (100%) |
|
|
355
388
|
|
|
@@ -358,11 +391,10 @@ pytest tests/evaluate/test_evaluator.py # Single module
|
|
|
358
391
|
### 🎨 Code Quality
|
|
359
392
|
|
|
360
393
|
```bash
|
|
361
|
-
pip install -e .[dev]
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
mypy dataflow
|
|
365
|
-
flake8 dataflow tests samples # Lint
|
|
394
|
+
pip install -e .[dev] # Install dev dependencies
|
|
395
|
+
ruff check dataflow tests samples # Lint
|
|
396
|
+
ruff format --check dataflow tests samples # Format check
|
|
397
|
+
mypy dataflow # Type check
|
|
366
398
|
```
|
|
367
399
|
|
|
368
400
|
### 🔗 Pre-commit Hooks (Optional)
|
|
@@ -372,7 +404,7 @@ pip install pre-commit
|
|
|
372
404
|
pre-commit install # Install git hooks (run once)
|
|
373
405
|
|
|
374
406
|
# After this, every `git commit` auto-runs:
|
|
375
|
-
#
|
|
407
|
+
# ruff (lint, auto-fix) → ruff format → whitespace checks
|
|
376
408
|
|
|
377
409
|
pre-commit run --all-files # Manual run against all files
|
|
378
410
|
```
|
|
@@ -388,7 +420,7 @@ dataflow/
|
|
|
388
420
|
├── evaluate/ # pycocotools-based metrics, log templates
|
|
389
421
|
├── util/ # Unified logging (LogManager + format helpers)
|
|
390
422
|
└── cli/ # CLI entry point, commands, validation
|
|
391
|
-
tests/ # Unit & integration tests (
|
|
423
|
+
tests/ # Unit & integration tests (606 tests, conftest fixtures)
|
|
392
424
|
samples/ # Python API usage examples (analyse, convert, visualize, evaluate, cli)
|
|
393
425
|
assets/ # Test data (det/seg by format)
|
|
394
426
|
specs/ # Canonical specifications (evaluate/ + formats/ + modules/)
|
|
@@ -8,7 +8,7 @@ import datetime
|
|
|
8
8
|
from abc import ABC, abstractmethod
|
|
9
9
|
from dataclasses import dataclass, field
|
|
10
10
|
from pathlib import Path
|
|
11
|
-
from typing import Any, Dict, List, Optional
|
|
11
|
+
from typing import Any, Dict, List, Optional, Set
|
|
12
12
|
|
|
13
13
|
from ..label.base import AnnotationResult, BaseAnnotationHandler
|
|
14
14
|
from ..label.models import AnnotationFormat, DatasetAnnotations, ImageAnnotation
|
|
@@ -215,6 +215,7 @@ class BaseConverter(ABC):
|
|
|
215
215
|
|
|
216
216
|
num_images = 0
|
|
217
217
|
num_objects = 0
|
|
218
|
+
written_ids: Set[str] = set() # Detect duplicate output stems
|
|
218
219
|
|
|
219
220
|
try:
|
|
220
221
|
# 1. Validate inputs
|
|
@@ -238,6 +239,38 @@ class BaseConverter(ABC):
|
|
|
238
239
|
|
|
239
240
|
for image_ann in source_handler.iter_images():
|
|
240
241
|
target_ann = self._convert_single_image(image_ann, **kwargs)
|
|
242
|
+
|
|
243
|
+
# Output-name notices: COCO-source converters rewrite
|
|
244
|
+
# image_id to the file_name stem (e.g. "1" → "000001"), and
|
|
245
|
+
# file_name directory components are dropped (per-file
|
|
246
|
+
# outputs are flat per the YOLO/LabelMe format specs).
|
|
247
|
+
renamed = target_ann.image_id != image_ann.image_id
|
|
248
|
+
flattened = self.source_format == "coco" and (
|
|
249
|
+
Path(str(image_ann.image_path).replace("\\", "/")).parent != Path(".")
|
|
250
|
+
)
|
|
251
|
+
if renamed or flattened:
|
|
252
|
+
parts = []
|
|
253
|
+
if renamed:
|
|
254
|
+
parts.append(f"renamed {image_ann.image_id} → {target_ann.image_id}")
|
|
255
|
+
if flattened:
|
|
256
|
+
parts.append(f"subdirectory flattened from '{image_ann.image_path}'")
|
|
257
|
+
notice = f"Output '{target_ann.image_id}': " + ", ".join(parts)
|
|
258
|
+
result.add_warning(notice)
|
|
259
|
+
self._log_warning(notice)
|
|
260
|
+
|
|
261
|
+
# Duplicate output stem detection: two source images sharing
|
|
262
|
+
# a basename (e.g. in different subdirectories) would silently
|
|
263
|
+
# overwrite each other's output file.
|
|
264
|
+
if target_ann.image_id in written_ids:
|
|
265
|
+
dup = (
|
|
266
|
+
f"Duplicate output stem '{target_ann.image_id}': "
|
|
267
|
+
f"{image_ann.image_path} overwrites a previously written file"
|
|
268
|
+
)
|
|
269
|
+
result.add_warning(dup)
|
|
270
|
+
self._log_warning(dup)
|
|
271
|
+
else:
|
|
272
|
+
written_ids.add(target_ann.image_id)
|
|
273
|
+
|
|
241
274
|
write_result = target_handler.write_one(target_ann, write_dir)
|
|
242
275
|
if not write_result.success:
|
|
243
276
|
err = f"Failed to write {target_ann.image_id}: {write_result.message}"
|
|
@@ -220,6 +220,16 @@ class CocoAndLabelMeConverter(BaseConverter):
|
|
|
220
220
|
Only the structural representation differs — coordinate values pass
|
|
221
221
|
through unchanged.
|
|
222
222
|
"""
|
|
223
|
+
# COCO → LabelMe only: rewrite image_id to the COCO file_name stem so
|
|
224
|
+
# the output .json shares the stem with the image file (LabelMe format
|
|
225
|
+
# requirement; leading zeros preserved). LabelMe → COCO must NOT
|
|
226
|
+
# rewrite — the COCO output id derives from the numeric image_id.
|
|
227
|
+
new_image_id = image_ann.image_id
|
|
228
|
+
if self.source_format == "coco":
|
|
229
|
+
from .utils import coco_file_name_to_image_id
|
|
230
|
+
|
|
231
|
+
new_image_id = coco_file_name_to_image_id(image_ann.image_path, image_ann.image_id)
|
|
232
|
+
|
|
223
233
|
new_objects = []
|
|
224
234
|
for obj in image_ann.objects:
|
|
225
235
|
new_bbox = None
|
|
@@ -251,7 +261,7 @@ class CocoAndLabelMeConverter(BaseConverter):
|
|
|
251
261
|
)
|
|
252
262
|
|
|
253
263
|
return ImageAnnotation(
|
|
254
|
-
image_id=
|
|
264
|
+
image_id=new_image_id,
|
|
255
265
|
image_path=image_ann.image_path,
|
|
256
266
|
width=image_ann.width,
|
|
257
267
|
height=image_ann.height,
|
|
@@ -65,9 +65,62 @@ def format_convert_phase(phase: str, stats: Dict[str, Any]) -> str:
|
|
|
65
65
|
return "\n".join(lines)
|
|
66
66
|
|
|
67
67
|
|
|
68
|
+
def format_output_layout(target_path: str) -> str:
|
|
69
|
+
"""Return a compact summary of the generated output layout.
|
|
70
|
+
|
|
71
|
+
Scans the actual target path and lists its top-level entries:
|
|
72
|
+
directories show their recursive file count (``labels/ 4 files``);
|
|
73
|
+
empty directories show ``empty`` — an empty ``images/`` additionally
|
|
74
|
+
notes that images are not copied; single-file targets (COCO JSON)
|
|
75
|
+
show the file name.
|
|
76
|
+
|
|
77
|
+
Args:
|
|
78
|
+
target_path: Target output path of a conversion.
|
|
79
|
+
|
|
80
|
+
Returns:
|
|
81
|
+
``"Output layout:"`` block string, or ``""`` when the target
|
|
82
|
+
path does not exist.
|
|
83
|
+
"""
|
|
84
|
+
from pathlib import Path
|
|
85
|
+
|
|
86
|
+
target = Path(target_path)
|
|
87
|
+
if not target.exists():
|
|
88
|
+
return ""
|
|
89
|
+
|
|
90
|
+
lines = ["Output layout:"]
|
|
91
|
+
if target.is_file():
|
|
92
|
+
lines.append(f" {target.name} (single file)")
|
|
93
|
+
return "\n".join(lines)
|
|
94
|
+
|
|
95
|
+
try:
|
|
96
|
+
entries = sorted(target.iterdir(), key=lambda p: (p.is_file(), p.name.lower()))
|
|
97
|
+
except OSError:
|
|
98
|
+
return ""
|
|
99
|
+
|
|
100
|
+
for entry in entries:
|
|
101
|
+
if entry.is_dir():
|
|
102
|
+
count = sum(1 for p in entry.rglob("*") if p.is_file())
|
|
103
|
+
if count == 0:
|
|
104
|
+
if entry.name == "images":
|
|
105
|
+
lines.append(
|
|
106
|
+
f" {entry.name}/ empty — images are not copied; "
|
|
107
|
+
"place your image files here"
|
|
108
|
+
)
|
|
109
|
+
else:
|
|
110
|
+
lines.append(f" {entry.name}/ empty")
|
|
111
|
+
else:
|
|
112
|
+
lines.append(f" {entry.name}/ {count} files")
|
|
113
|
+
else:
|
|
114
|
+
lines.append(f" {entry.name}")
|
|
115
|
+
return "\n".join(lines)
|
|
116
|
+
|
|
117
|
+
|
|
68
118
|
def format_convert_result(result: Any) -> str:
|
|
69
119
|
"""Return a final result block for a completed conversion.
|
|
70
120
|
|
|
121
|
+
The block is followed by an output layout summary (see
|
|
122
|
+
``format_output_layout()``) when the target path exists.
|
|
123
|
+
|
|
71
124
|
Args:
|
|
72
125
|
result: A ``ConversionResult`` instance.
|
|
73
126
|
|
|
@@ -90,4 +143,11 @@ def format_convert_result(result: Any) -> str:
|
|
|
90
143
|
if result.warnings:
|
|
91
144
|
items["Warnings"] = len(result.warnings)
|
|
92
145
|
|
|
93
|
-
|
|
146
|
+
block = format_result_block(status, items, log_path=result.log_path)
|
|
147
|
+
|
|
148
|
+
if getattr(result, "target_path", None):
|
|
149
|
+
layout = format_output_layout(result.target_path)
|
|
150
|
+
if layout:
|
|
151
|
+
block = f"{block}\n\n{layout}"
|
|
152
|
+
|
|
153
|
+
return block
|
|
@@ -130,6 +130,42 @@ def absolute_pixel_to_yolo(
|
|
|
130
130
|
# ---------------------------------------------------------------------------
|
|
131
131
|
|
|
132
132
|
|
|
133
|
+
def coco_file_name_to_image_id(file_name: Optional[str], fallback_id: str) -> str:
|
|
134
|
+
"""Derive the per-file output ``image_id`` from a COCO ``file_name``.
|
|
135
|
+
|
|
136
|
+
COCO ``file_name`` is the authoritative source of the image file stem
|
|
137
|
+
(``id`` is only a numeric reference key and carries no filename
|
|
138
|
+
information). Per-file output formats (YOLO `.txt`, LabelMe `.json`)
|
|
139
|
+
must share the stem with the image file (see `spec_yolo_format.md` /
|
|
140
|
+
`spec_labelme_format.md`), so COCO-source converters rewrite ``image_id``
|
|
141
|
+
to this stem before ``write_one()``.
|
|
142
|
+
|
|
143
|
+
Rules:
|
|
144
|
+
- Backslashes are normalized to ``/`` (Windows-style paths)
|
|
145
|
+
- Subdirectory components are dropped (output is flat; the
|
|
146
|
+
`spec_label.md` image_id invariant forbids path separators)
|
|
147
|
+
- Leading zeros are preserved (``000001.jpg`` → ``000001``)
|
|
148
|
+
- Degenerate stems (``""``, ``"."``, ``".."``) fall back to *fallback_id*
|
|
149
|
+
(the numeric COCO id string)
|
|
150
|
+
|
|
151
|
+
Args:
|
|
152
|
+
file_name: COCO ``images[].file_name`` value. ``None`` is tolerated
|
|
153
|
+
defensively and treated as empty (falls back to *fallback_id*).
|
|
154
|
+
fallback_id: ``ImageAnnotation.image_id`` from the COCO handler
|
|
155
|
+
(numeric id string), used when the stem is degenerate.
|
|
156
|
+
|
|
157
|
+
Returns:
|
|
158
|
+
Basename stem of *file_name*, or *fallback_id*.
|
|
159
|
+
"""
|
|
160
|
+
if not file_name:
|
|
161
|
+
return fallback_id
|
|
162
|
+
normalized = str(file_name).replace("\\", "/")
|
|
163
|
+
stem = Path(normalized).stem
|
|
164
|
+
if stem in ("", ".", ".."):
|
|
165
|
+
return fallback_id
|
|
166
|
+
return stem
|
|
167
|
+
|
|
168
|
+
|
|
133
169
|
def ensure_coco_categories_for_streaming(
|
|
134
170
|
converter: Any,
|
|
135
171
|
source_handler: Any,
|
|
@@ -288,7 +288,7 @@ class YoloAndCocoConverter(BaseConverter):
|
|
|
288
288
|
|
|
289
289
|
def _coco_to_yolo_one(self, img: ImageAnnotation) -> ImageAnnotation:
|
|
290
290
|
"""Convert single image: COCO absolute px → YOLO normalized center."""
|
|
291
|
-
from .utils import absolute_pixel_to_yolo
|
|
291
|
+
from .utils import absolute_pixel_to_yolo, coco_file_name_to_image_id
|
|
292
292
|
|
|
293
293
|
new_objects = []
|
|
294
294
|
for obj in img.objects:
|
|
@@ -306,8 +306,11 @@ class YoloAndCocoConverter(BaseConverter):
|
|
|
306
306
|
)
|
|
307
307
|
)
|
|
308
308
|
|
|
309
|
+
# YOLO format requires label files to share the stem with the image
|
|
310
|
+
# file — derive the output name from the COCO file_name stem (which
|
|
311
|
+
# preserves leading zeros), not from the numeric COCO id.
|
|
309
312
|
return ImageAnnotation(
|
|
310
|
-
image_id=img.image_id,
|
|
313
|
+
image_id=coco_file_name_to_image_id(img.image_path, img.image_id),
|
|
311
314
|
image_path=img.image_path,
|
|
312
315
|
width=img.width,
|
|
313
316
|
height=img.height,
|
|
@@ -203,11 +203,17 @@ def _dataset_to_coco_dict(dataset: Any) -> Dict[str, Any]:
|
|
|
203
203
|
# Preserve top-level keys from dataset_info
|
|
204
204
|
info = dataset.dataset_info.copy()
|
|
205
205
|
|
|
206
|
-
# Reconstruct images
|
|
206
|
+
# Reconstruct images — image ids resolve with the same rule as the COCO
|
|
207
|
+
# handler write() (digit image_id → int, non-digit/colliding → dedicated
|
|
208
|
+
# counter, unique ids guaranteed — resolve_coco_image_ids()).
|
|
209
|
+
from dataflow.label.utils import resolve_coco_image_ids
|
|
210
|
+
|
|
211
|
+
id_map = resolve_coco_image_ids(img.image_id for img in dataset.images)
|
|
212
|
+
|
|
207
213
|
images = []
|
|
208
214
|
for img in dataset.images:
|
|
209
215
|
img_entry = {
|
|
210
|
-
"id":
|
|
216
|
+
"id": id_map[img.image_id],
|
|
211
217
|
"file_name": img.image_path,
|
|
212
218
|
"width": img.width,
|
|
213
219
|
"height": img.height,
|
|
@@ -223,7 +229,7 @@ def _dataset_to_coco_dict(dataset: Any) -> Dict[str, Any]:
|
|
|
223
229
|
annotations = []
|
|
224
230
|
ann_id = 1
|
|
225
231
|
for img in dataset.images:
|
|
226
|
-
image_id =
|
|
232
|
+
image_id = id_map[img.image_id]
|
|
227
233
|
for obj in img.objects:
|
|
228
234
|
ann_entry: Dict[str, Any] = {
|
|
229
235
|
"id": ann_id,
|
|
@@ -607,12 +607,18 @@ class CocoAnnotationHandler(BaseAnnotationHandler):
|
|
|
607
607
|
images = []
|
|
608
608
|
coco_annotations = []
|
|
609
609
|
ann_id = 1
|
|
610
|
-
|
|
610
|
+
|
|
611
|
+
# Resolve format-native image_id strings to unique positive-int COCO
|
|
612
|
+
# ids (digit strings keep their value; non-digit and colliding ids
|
|
613
|
+
# get dedicated counter ids — see resolve_coco_image_ids()).
|
|
614
|
+
from .utils import resolve_coco_image_ids
|
|
615
|
+
|
|
616
|
+
id_map = resolve_coco_image_ids(img.image_id for img in annotations.images)
|
|
611
617
|
|
|
612
618
|
for img in annotations.images:
|
|
613
619
|
images.append(
|
|
614
620
|
{
|
|
615
|
-
"id":
|
|
621
|
+
"id": id_map[img.image_id],
|
|
616
622
|
"width": img.width,
|
|
617
623
|
"height": img.height,
|
|
618
624
|
"file_name": img.image_path,
|
|
@@ -625,7 +631,7 @@ class CocoAnnotationHandler(BaseAnnotationHandler):
|
|
|
625
631
|
|
|
626
632
|
# Add object annotations
|
|
627
633
|
for obj in img.objects:
|
|
628
|
-
coco_ann = self._object_to_coco_annotation(obj, img, ann_id,
|
|
634
|
+
coco_ann = self._object_to_coco_annotation(obj, img, ann_id, id_map[img.image_id])
|
|
629
635
|
if coco_ann:
|
|
630
636
|
coco_annotations.append(coco_ann)
|
|
631
637
|
ann_id += 1
|
|
@@ -639,8 +645,6 @@ class CocoAnnotationHandler(BaseAnnotationHandler):
|
|
|
639
645
|
f"Skipping object {obj.class_name}: conversion to COCO format failed"
|
|
640
646
|
)
|
|
641
647
|
|
|
642
|
-
img_id_counter += 1
|
|
643
|
-
|
|
644
648
|
# Prediction mode: output plain list of annotation dicts
|
|
645
649
|
if self.prediction:
|
|
646
650
|
return coco_annotations
|
|
@@ -661,9 +665,14 @@ class CocoAnnotationHandler(BaseAnnotationHandler):
|
|
|
661
665
|
return result
|
|
662
666
|
|
|
663
667
|
def _object_to_coco_annotation(
|
|
664
|
-
self, obj: ObjectAnnotation, img: ImageAnnotation, ann_id: int,
|
|
668
|
+
self, obj: ObjectAnnotation, img: ImageAnnotation, ann_id: int, resolved_img_id: int
|
|
665
669
|
) -> Optional[Dict]:
|
|
666
|
-
"""Convert ObjectAnnotation to COCO annotation dict.
|
|
670
|
+
"""Convert ObjectAnnotation to COCO annotation dict.
|
|
671
|
+
|
|
672
|
+
Args:
|
|
673
|
+
resolved_img_id: COCO image id resolved from ``img.image_id`` by
|
|
674
|
+
``resolve_coco_image_ids()`` (see ``_prepare_coco_data``).
|
|
675
|
+
"""
|
|
667
676
|
try:
|
|
668
677
|
# Determine segmentation format
|
|
669
678
|
segmentation = None
|
|
@@ -767,10 +776,8 @@ class CocoAnnotationHandler(BaseAnnotationHandler):
|
|
|
767
776
|
bbox = [float(min_x), float(min_y), float(w), float(h)]
|
|
768
777
|
area = float(w * h)
|
|
769
778
|
|
|
770
|
-
image_id_val = int(img.image_id) if img.image_id.isdigit() else img_id
|
|
771
|
-
|
|
772
779
|
ann_dict = {
|
|
773
|
-
"image_id":
|
|
780
|
+
"image_id": resolved_img_id,
|
|
774
781
|
"category_id": obj.class_id,
|
|
775
782
|
"segmentation": segmentation,
|
|
776
783
|
"area": area,
|
|
@@ -4,7 +4,52 @@ Utility functions for the label module.
|
|
|
4
4
|
|
|
5
5
|
import hashlib
|
|
6
6
|
from pathlib import Path
|
|
7
|
-
from typing import Dict, Optional
|
|
7
|
+
from typing import Dict, Iterable, Optional, Set
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def resolve_coco_image_ids(image_ids: Iterable[str]) -> Dict[str, int]:
|
|
11
|
+
"""Resolve format-native ``image_id`` strings to unique positive-int COCO ids.
|
|
12
|
+
|
|
13
|
+
COCO ``images[].id`` must be an int and unique per image
|
|
14
|
+
(`spec_coco_format.md`). ``ImageAnnotation.image_id`` is a string and may
|
|
15
|
+
be a file stem, so COCO writers resolve it:
|
|
16
|
+
|
|
17
|
+
- Digit-string ``image_id`` → its ``int`` value (preserves roundtrip ids)
|
|
18
|
+
- Non-digit ``image_id`` → a dedicated counter id (not the annotation
|
|
19
|
+
counter); counters skip values reserved by digit-string ids
|
|
20
|
+
- Uniqueness guaranteed: if a resolved id collides (e.g. ``"01"`` vs
|
|
21
|
+
``"1"``), the later image gets a fresh counter id
|
|
22
|
+
|
|
23
|
+
Canonical implementation — used by both
|
|
24
|
+
``CocoAnnotationHandler._prepare_coco_data()`` and
|
|
25
|
+
``evaluate/utils._dataset_to_coco_dict()``.
|
|
26
|
+
|
|
27
|
+
Args:
|
|
28
|
+
image_ids: Iterable of image_id strings in output order.
|
|
29
|
+
|
|
30
|
+
Returns:
|
|
31
|
+
Mapping ``{image_id_string: resolved_int_id}``.
|
|
32
|
+
"""
|
|
33
|
+
ids = [str(s) for s in image_ids]
|
|
34
|
+
# Values reserved by digit-string ids — counters must never take them,
|
|
35
|
+
# otherwise a later digit id would collide with an earlier counter.
|
|
36
|
+
digit_values: Set[int] = {int(s) for s in ids if s.isdigit()}
|
|
37
|
+
used: Set[int] = set()
|
|
38
|
+
mapping: Dict[str, int] = {}
|
|
39
|
+
counter = 1
|
|
40
|
+
|
|
41
|
+
for s in ids:
|
|
42
|
+
if s.isdigit() and int(s) not in used:
|
|
43
|
+
value = int(s)
|
|
44
|
+
else:
|
|
45
|
+
while counter in used or counter in digit_values:
|
|
46
|
+
counter += 1
|
|
47
|
+
value = counter
|
|
48
|
+
counter += 1
|
|
49
|
+
used.add(value)
|
|
50
|
+
mapping[s] = value
|
|
51
|
+
|
|
52
|
+
return mapping
|
|
8
53
|
|
|
9
54
|
|
|
10
55
|
def parse_yolo_class_id(token: str) -> Optional[int]:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dataflow-cv
|
|
3
|
-
Version:
|
|
3
|
+
Version: 3.0.0
|
|
4
4
|
Summary: A computer vision dataset processing library — analyse, convert, visualize, and evaluate annotations across YOLO, LabelMe, and COCO formats
|
|
5
5
|
Author: DataFlow-CV Team
|
|
6
6
|
License: MIT
|
|
@@ -156,6 +156,12 @@ dataflow-cv convert yolo2coco --verbose images/ labels/ classes.txt output.json
|
|
|
156
156
|
dataflow-cv convert yolo2coco --no-strict images/ labels/ classes.txt output.json
|
|
157
157
|
```
|
|
158
158
|
|
|
159
|
+
> 📂 **COCO→YOLO / COCO→LabelMe output naming**: each `.txt` / `.json` is named by the
|
|
160
|
+
> image file stem — leading zeros preserved (`000001.jpg` → `000001.txt`). The
|
|
161
|
+
> converter generates `labels/` (or the `.json` files) + `classes.txt`; the
|
|
162
|
+
> `images/` directory is created but left **empty** — images are never copied,
|
|
163
|
+
> place your image files there yourself.
|
|
164
|
+
|
|
159
165
|
#### 🎨 Visualization
|
|
160
166
|
|
|
161
167
|
```bash
|
|
@@ -231,7 +237,13 @@ Two evaluation modes, distinguished by how overlap is measured:
|
|
|
231
237
|
|
|
232
238
|
```python
|
|
233
239
|
from dataflow.util.logging import LogConfig
|
|
234
|
-
from dataflow.analyse import
|
|
240
|
+
from dataflow.analyse import (
|
|
241
|
+
StatsAnalyser,
|
|
242
|
+
SplitAnalyser,
|
|
243
|
+
FilterAnalyser,
|
|
244
|
+
PartitionAnalyser,
|
|
245
|
+
SampleAnalyser,
|
|
246
|
+
)
|
|
235
247
|
from dataflow.convert import YoloAndCocoConverter
|
|
236
248
|
from dataflow.visualize import YOLOVisualizer
|
|
237
249
|
from dataflow.evaluate import DetectionEvaluator, compute_pr_f1
|
|
@@ -247,37 +259,50 @@ print(f"{result.data.total_files} images, {result.data.total_annotations} object
|
|
|
247
259
|
# Train/test split (YOLO / LabelMe)
|
|
248
260
|
splitter = SplitAnalyser(log_config=log_cfg)
|
|
249
261
|
result = splitter.analyse(
|
|
250
|
-
output_dir="output/",
|
|
251
|
-
|
|
262
|
+
output_dir="output/",
|
|
263
|
+
ratio=0.8,
|
|
264
|
+
seed=42,
|
|
265
|
+
label_dir="yolo_labels/",
|
|
266
|
+
class_file="classes.txt",
|
|
252
267
|
)
|
|
253
268
|
print(f"Train: {result.data.train_count}, Val: {result.data.val_count}")
|
|
254
269
|
|
|
255
270
|
# Split with images (both mode — labels drive, images follow by stem)
|
|
256
271
|
result = splitter.analyse(
|
|
257
|
-
output_dir="output/",
|
|
258
|
-
|
|
272
|
+
output_dir="output/",
|
|
273
|
+
ratio=0.8,
|
|
274
|
+
seed=42,
|
|
275
|
+
label_dir="yolo_labels/",
|
|
276
|
+
image_dir="images/",
|
|
259
277
|
class_file="classes.txt",
|
|
260
278
|
)
|
|
261
279
|
|
|
262
280
|
# Category filter (keep / remap categories per new classes.txt)
|
|
263
281
|
filterer = FilterAnalyser(log_config=log_cfg)
|
|
264
282
|
result = filterer.analyse(
|
|
265
|
-
"yolo_labels/",
|
|
266
|
-
|
|
283
|
+
"yolo_labels/",
|
|
284
|
+
original_class_file="classes.txt",
|
|
285
|
+
new_class_file="classes_new.txt",
|
|
286
|
+
output_dir="filtered/",
|
|
267
287
|
)
|
|
268
288
|
|
|
269
289
|
# N-way partition (YOLO / LabelMe labels; images follow by stem)
|
|
270
290
|
partitioner = PartitionAnalyser(log_config=log_cfg)
|
|
271
291
|
result = partitioner.analyse(
|
|
272
|
-
output_dir="parts/",
|
|
273
|
-
|
|
292
|
+
output_dir="parts/",
|
|
293
|
+
num=4,
|
|
294
|
+
label_dir="yolo_labels/",
|
|
295
|
+
image_dir="images/",
|
|
274
296
|
)
|
|
275
297
|
|
|
276
298
|
# File sampling (labels, images, or both — random or sequential)
|
|
277
299
|
sampler = SampleAnalyser(log_config=log_cfg)
|
|
278
300
|
result = sampler.analyse(
|
|
279
|
-
output_dir="sampled/",
|
|
280
|
-
|
|
301
|
+
output_dir="sampled/",
|
|
302
|
+
count=10,
|
|
303
|
+
label_dir="yolo_labels/",
|
|
304
|
+
shuffle=True,
|
|
305
|
+
seed=42,
|
|
281
306
|
)
|
|
282
307
|
|
|
283
308
|
# ── Convert ──────────────────────────────────────────
|
|
@@ -285,22 +310,30 @@ result = sampler.analyse(
|
|
|
285
310
|
log_cfg = LogConfig(name="convert", verbose=True)
|
|
286
311
|
converter = YoloAndCocoConverter(source_to_target=True, log_config=log_cfg, strict_mode=True)
|
|
287
312
|
result = converter.convert(
|
|
288
|
-
source_path="yolo_labels/",
|
|
289
|
-
|
|
313
|
+
source_path="yolo_labels/",
|
|
314
|
+
target_path="anno.json",
|
|
315
|
+
class_file="classes.txt",
|
|
316
|
+
image_dir="images/",
|
|
290
317
|
)
|
|
291
318
|
|
|
292
319
|
# YOLO predictions → COCO (prediction mode)
|
|
293
320
|
converter = YoloAndCocoConverter(source_to_target=True, prediction=True)
|
|
294
321
|
result = converter.convert(
|
|
295
|
-
source_path="yolo_preds/",
|
|
296
|
-
|
|
322
|
+
source_path="yolo_preds/",
|
|
323
|
+
target_path="pred.json",
|
|
324
|
+
class_file="classes.txt",
|
|
325
|
+
image_dir="images/",
|
|
297
326
|
)
|
|
298
327
|
|
|
299
328
|
# ── Visualize ────────────────────────────────────────
|
|
300
329
|
visualizer = YOLOVisualizer(
|
|
301
|
-
label_dir="yolo_labels/",
|
|
302
|
-
|
|
303
|
-
|
|
330
|
+
label_dir="yolo_labels/",
|
|
331
|
+
image_dir="images/",
|
|
332
|
+
class_file="classes.txt",
|
|
333
|
+
is_show=True,
|
|
334
|
+
is_save=True,
|
|
335
|
+
output_dir="visualized/",
|
|
336
|
+
log_config=log_cfg,
|
|
304
337
|
)
|
|
305
338
|
result = visualizer.visualize()
|
|
306
339
|
|
|
@@ -370,7 +403,7 @@ For detailed developer guidance including advanced test commands, debugging, and
|
|
|
370
403
|
|
|
371
404
|
### 🧪 Testing
|
|
372
405
|
|
|
373
|
-
**
|
|
406
|
+
**606 tests, 80% code coverage (5532 statements).**
|
|
374
407
|
|
|
375
408
|
```bash
|
|
376
409
|
pytest # All tests
|
|
@@ -384,11 +417,11 @@ pytest tests/evaluate/test_evaluator.py # Single module
|
|
|
384
417
|
|
|
385
418
|
| Module | Coverage | Highlights |
|
|
386
419
|
|--------|:--------:|------------|
|
|
387
|
-
| `dataflow/label/` | 71% | models (84%), base (82%), utils (
|
|
388
|
-
| `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (
|
|
389
|
-
| `dataflow/convert/` |
|
|
420
|
+
| `dataflow/label/` | 71% | models (84%), base (82%), utils (83%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
|
|
421
|
+
| `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (84%), split (85%), stats (82%), filter (76%), partition (74%) |
|
|
422
|
+
| `dataflow/convert/` | 86% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (88%), coco_and_labelme (88%), log_templates (85%), base (81%), rle (80%) |
|
|
390
423
|
| `dataflow/visualize/` | 81% | yolo_vis (100%), labelme_vis (100%), coco_vis (93%), base (76%) |
|
|
391
|
-
| `dataflow/evaluate/` |
|
|
424
|
+
| `dataflow/evaluate/` | 90% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (87%) |
|
|
392
425
|
| `dataflow/cli/` | 74% | main (96%), visualize cmd (87%), utils (87%), evaluate cmd (83%), analyse cmd (65%), convert cmd (52%) |
|
|
393
426
|
| `dataflow/util/` | 100% | logging (100%) |
|
|
394
427
|
|
|
@@ -397,11 +430,10 @@ pytest tests/evaluate/test_evaluator.py # Single module
|
|
|
397
430
|
### 🎨 Code Quality
|
|
398
431
|
|
|
399
432
|
```bash
|
|
400
|
-
pip install -e .[dev]
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
mypy dataflow
|
|
404
|
-
flake8 dataflow tests samples # Lint
|
|
433
|
+
pip install -e .[dev] # Install dev dependencies
|
|
434
|
+
ruff check dataflow tests samples # Lint
|
|
435
|
+
ruff format --check dataflow tests samples # Format check
|
|
436
|
+
mypy dataflow # Type check
|
|
405
437
|
```
|
|
406
438
|
|
|
407
439
|
### 🔗 Pre-commit Hooks (Optional)
|
|
@@ -411,7 +443,7 @@ pip install pre-commit
|
|
|
411
443
|
pre-commit install # Install git hooks (run once)
|
|
412
444
|
|
|
413
445
|
# After this, every `git commit` auto-runs:
|
|
414
|
-
#
|
|
446
|
+
# ruff (lint, auto-fix) → ruff format → whitespace checks
|
|
415
447
|
|
|
416
448
|
pre-commit run --all-files # Manual run against all files
|
|
417
449
|
```
|
|
@@ -427,7 +459,7 @@ dataflow/
|
|
|
427
459
|
├── evaluate/ # pycocotools-based metrics, log templates
|
|
428
460
|
├── util/ # Unified logging (LogManager + format helpers)
|
|
429
461
|
└── cli/ # CLI entry point, commands, validation
|
|
430
|
-
tests/ # Unit & integration tests (
|
|
462
|
+
tests/ # Unit & integration tests (606 tests, conftest fixtures)
|
|
431
463
|
samples/ # Python API usage examples (analyse, convert, visualize, evaluate, cli)
|
|
432
464
|
assets/ # Test data (det/seg by format)
|
|
433
465
|
specs/ # Canonical specifications (evaluate/ + formats/ + modules/)
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dataflow-cv"
|
|
7
|
-
version = "
|
|
7
|
+
version = "3.0.0"
|
|
8
8
|
description = "A computer vision dataset processing library — analyse, convert, visualize, and evaluate annotations across YOLO, LabelMe, and COCO formats"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.8"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|