dataflow-cv 2.0.0__tar.gz → 3.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dataflow_cv-2.0.0/dataflow_cv.egg-info → dataflow_cv-3.0.0}/PKG-INFO +65 -34
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/README.md +63 -30
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/__init__.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/filter.py +30 -58
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/log_templates.py +27 -29
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/partition.py +17 -46
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/sample.py +8 -23
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/split.py +8 -27
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/stats.py +23 -21
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/utils.py +12 -31
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/__init__.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/__init__.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/analyse.py +18 -34
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/convert.py +28 -15
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/evaluate.py +16 -4
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/utils.py +12 -10
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/commands/visualize.py +15 -7
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/exceptions.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/cli/main.py +14 -9
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/base.py +81 -73
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/coco_and_labelme.py +42 -31
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/labelme_and_yolo.py +30 -40
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/log_templates.py +61 -1
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/rle_converter.py +3 -10
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/utils.py +52 -42
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/yolo_and_coco.py +38 -46
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/base.py +6 -16
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/evaluator.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/log_templates.py +40 -14
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/metrics.py +28 -32
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/result.py +23 -26
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/utils.py +21 -34
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/__init__.py +8 -2
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/base.py +20 -48
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/coco_handler.py +65 -92
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/labelme_handler.py +51 -99
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/models.py +7 -21
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/utils.py +48 -7
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/label/yolo_handler.py +78 -143
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/util/logging.py +8 -28
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/__init__.py +1 -2
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/base.py +21 -60
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/coco_visualizer.py +4 -10
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/labelme_visualizer.py +2 -7
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/log_templates.py +1 -1
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/utils.py +0 -1
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/visualize/yolo_visualizer.py +4 -12
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0/dataflow_cv.egg-info}/PKG-INFO +65 -34
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/requires.txt +1 -3
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/pyproject.toml +10 -26
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/LICENSE +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/__init__.py +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/analyse/base.py +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/convert/__init__.py +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/evaluate/__init__.py +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow/util/__init__.py +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/SOURCES.txt +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/dependency_links.txt +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/entry_points.txt +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/not-zip-safe +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/dataflow_cv.egg-info/top_level.txt +0 -0
- {dataflow_cv-2.0.0 → dataflow_cv-3.0.0}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dataflow-cv
|
|
3
|
-
Version:
|
|
3
|
+
Version: 3.0.0
|
|
4
4
|
Summary: A computer vision dataset processing library — analyse, convert, visualize, and evaluate annotations across YOLO, LabelMe, and COCO formats
|
|
5
5
|
Author: DataFlow-CV Team
|
|
6
6
|
License: MIT
|
|
@@ -33,9 +33,7 @@ Requires-Dist: pycocotools>=2.0.0; extra == "coco"
|
|
|
33
33
|
Provides-Extra: dev
|
|
34
34
|
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
35
35
|
Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
|
|
36
|
-
Requires-Dist:
|
|
37
|
-
Requires-Dist: isort>=5.12.0; extra == "dev"
|
|
38
|
-
Requires-Dist: flake8>=6.0.0; extra == "dev"
|
|
36
|
+
Requires-Dist: ruff>=0.16; extra == "dev"
|
|
39
37
|
Requires-Dist: mypy>=1.0.0; extra == "dev"
|
|
40
38
|
Dynamic: license-file
|
|
41
39
|
|
|
@@ -67,6 +65,7 @@ A computer vision dataset processing library — analyse, convert, visualize, an
|
|
|
67
65
|
| 🎨 **Visualize** | OpenCV rendering with color-coded classes, display & save modes | `dataflow-cv visualize yolo ...` |
|
|
68
66
|
| 📊 **Evaluate** | COCO mAP via pycocotools, single-threshold P/R/F1 per class | `dataflow-cv evaluate detection ...` |
|
|
69
67
|
| 💻 **CLI + API** | Click-based CLI with rich `--help`; Python API for pipelines | `from dataflow.convert import ...` |
|
|
68
|
+
| 🤖 **AI Skills** | Claude Code skill (`/dataflow:dataflow-cv`) for AI assistants — CLI/API reference & known gotchas | `claude plugin install dataflow@claude-skills` |
|
|
70
69
|
|
|
71
70
|
---
|
|
72
71
|
|
|
@@ -157,6 +156,12 @@ dataflow-cv convert yolo2coco --verbose images/ labels/ classes.txt output.json
|
|
|
157
156
|
dataflow-cv convert yolo2coco --no-strict images/ labels/ classes.txt output.json
|
|
158
157
|
```
|
|
159
158
|
|
|
159
|
+
> 📂 **COCO→YOLO / COCO→LabelMe output naming**: each `.txt` / `.json` is named by the
|
|
160
|
+
> image file stem — leading zeros preserved (`000001.jpg` → `000001.txt`). The
|
|
161
|
+
> converter generates `labels/` (or the `.json` files) + `classes.txt`; the
|
|
162
|
+
> `images/` directory is created but left **empty** — images are never copied,
|
|
163
|
+
> place your image files there yourself.
|
|
164
|
+
|
|
160
165
|
#### 🎨 Visualization
|
|
161
166
|
|
|
162
167
|
```bash
|
|
@@ -232,7 +237,13 @@ Two evaluation modes, distinguished by how overlap is measured:
|
|
|
232
237
|
|
|
233
238
|
```python
|
|
234
239
|
from dataflow.util.logging import LogConfig
|
|
235
|
-
from dataflow.analyse import
|
|
240
|
+
from dataflow.analyse import (
|
|
241
|
+
StatsAnalyser,
|
|
242
|
+
SplitAnalyser,
|
|
243
|
+
FilterAnalyser,
|
|
244
|
+
PartitionAnalyser,
|
|
245
|
+
SampleAnalyser,
|
|
246
|
+
)
|
|
236
247
|
from dataflow.convert import YoloAndCocoConverter
|
|
237
248
|
from dataflow.visualize import YOLOVisualizer
|
|
238
249
|
from dataflow.evaluate import DetectionEvaluator, compute_pr_f1
|
|
@@ -248,37 +259,50 @@ print(f"{result.data.total_files} images, {result.data.total_annotations} object
|
|
|
248
259
|
# Train/test split (YOLO / LabelMe)
|
|
249
260
|
splitter = SplitAnalyser(log_config=log_cfg)
|
|
250
261
|
result = splitter.analyse(
|
|
251
|
-
output_dir="output/",
|
|
252
|
-
|
|
262
|
+
output_dir="output/",
|
|
263
|
+
ratio=0.8,
|
|
264
|
+
seed=42,
|
|
265
|
+
label_dir="yolo_labels/",
|
|
266
|
+
class_file="classes.txt",
|
|
253
267
|
)
|
|
254
268
|
print(f"Train: {result.data.train_count}, Val: {result.data.val_count}")
|
|
255
269
|
|
|
256
270
|
# Split with images (both mode — labels drive, images follow by stem)
|
|
257
271
|
result = splitter.analyse(
|
|
258
|
-
output_dir="output/",
|
|
259
|
-
|
|
272
|
+
output_dir="output/",
|
|
273
|
+
ratio=0.8,
|
|
274
|
+
seed=42,
|
|
275
|
+
label_dir="yolo_labels/",
|
|
276
|
+
image_dir="images/",
|
|
260
277
|
class_file="classes.txt",
|
|
261
278
|
)
|
|
262
279
|
|
|
263
280
|
# Category filter (keep / remap categories per new classes.txt)
|
|
264
281
|
filterer = FilterAnalyser(log_config=log_cfg)
|
|
265
282
|
result = filterer.analyse(
|
|
266
|
-
"yolo_labels/",
|
|
267
|
-
|
|
283
|
+
"yolo_labels/",
|
|
284
|
+
original_class_file="classes.txt",
|
|
285
|
+
new_class_file="classes_new.txt",
|
|
286
|
+
output_dir="filtered/",
|
|
268
287
|
)
|
|
269
288
|
|
|
270
289
|
# N-way partition (YOLO / LabelMe labels; images follow by stem)
|
|
271
290
|
partitioner = PartitionAnalyser(log_config=log_cfg)
|
|
272
291
|
result = partitioner.analyse(
|
|
273
|
-
output_dir="parts/",
|
|
274
|
-
|
|
292
|
+
output_dir="parts/",
|
|
293
|
+
num=4,
|
|
294
|
+
label_dir="yolo_labels/",
|
|
295
|
+
image_dir="images/",
|
|
275
296
|
)
|
|
276
297
|
|
|
277
298
|
# File sampling (labels, images, or both — random or sequential)
|
|
278
299
|
sampler = SampleAnalyser(log_config=log_cfg)
|
|
279
300
|
result = sampler.analyse(
|
|
280
|
-
output_dir="sampled/",
|
|
281
|
-
|
|
301
|
+
output_dir="sampled/",
|
|
302
|
+
count=10,
|
|
303
|
+
label_dir="yolo_labels/",
|
|
304
|
+
shuffle=True,
|
|
305
|
+
seed=42,
|
|
282
306
|
)
|
|
283
307
|
|
|
284
308
|
# ── Convert ──────────────────────────────────────────
|
|
@@ -286,22 +310,30 @@ result = sampler.analyse(
|
|
|
286
310
|
log_cfg = LogConfig(name="convert", verbose=True)
|
|
287
311
|
converter = YoloAndCocoConverter(source_to_target=True, log_config=log_cfg, strict_mode=True)
|
|
288
312
|
result = converter.convert(
|
|
289
|
-
source_path="yolo_labels/",
|
|
290
|
-
|
|
313
|
+
source_path="yolo_labels/",
|
|
314
|
+
target_path="anno.json",
|
|
315
|
+
class_file="classes.txt",
|
|
316
|
+
image_dir="images/",
|
|
291
317
|
)
|
|
292
318
|
|
|
293
319
|
# YOLO predictions → COCO (prediction mode)
|
|
294
320
|
converter = YoloAndCocoConverter(source_to_target=True, prediction=True)
|
|
295
321
|
result = converter.convert(
|
|
296
|
-
source_path="yolo_preds/",
|
|
297
|
-
|
|
322
|
+
source_path="yolo_preds/",
|
|
323
|
+
target_path="pred.json",
|
|
324
|
+
class_file="classes.txt",
|
|
325
|
+
image_dir="images/",
|
|
298
326
|
)
|
|
299
327
|
|
|
300
328
|
# ── Visualize ────────────────────────────────────────
|
|
301
329
|
visualizer = YOLOVisualizer(
|
|
302
|
-
label_dir="yolo_labels/",
|
|
303
|
-
|
|
304
|
-
|
|
330
|
+
label_dir="yolo_labels/",
|
|
331
|
+
image_dir="images/",
|
|
332
|
+
class_file="classes.txt",
|
|
333
|
+
is_show=True,
|
|
334
|
+
is_save=True,
|
|
335
|
+
output_dir="visualized/",
|
|
336
|
+
log_config=log_cfg,
|
|
305
337
|
)
|
|
306
338
|
result = visualizer.visualize()
|
|
307
339
|
|
|
@@ -371,7 +403,7 @@ For detailed developer guidance including advanced test commands, debugging, and
|
|
|
371
403
|
|
|
372
404
|
### 🧪 Testing
|
|
373
405
|
|
|
374
|
-
**
|
|
406
|
+
**606 tests, 80% code coverage (5532 statements).**
|
|
375
407
|
|
|
376
408
|
```bash
|
|
377
409
|
pytest # All tests
|
|
@@ -385,11 +417,11 @@ pytest tests/evaluate/test_evaluator.py # Single module
|
|
|
385
417
|
|
|
386
418
|
| Module | Coverage | Highlights |
|
|
387
419
|
|--------|:--------:|------------|
|
|
388
|
-
| `dataflow/label/` | 71% | models (84%), base (82%), utils (
|
|
389
|
-
| `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (
|
|
390
|
-
| `dataflow/convert/` |
|
|
420
|
+
| `dataflow/label/` | 71% | models (84%), base (82%), utils (83%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
|
|
421
|
+
| `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (84%), split (85%), stats (82%), filter (76%), partition (74%) |
|
|
422
|
+
| `dataflow/convert/` | 86% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (88%), coco_and_labelme (88%), log_templates (85%), base (81%), rle (80%) |
|
|
391
423
|
| `dataflow/visualize/` | 81% | yolo_vis (100%), labelme_vis (100%), coco_vis (93%), base (76%) |
|
|
392
|
-
| `dataflow/evaluate/` |
|
|
424
|
+
| `dataflow/evaluate/` | 90% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (87%) |
|
|
393
425
|
| `dataflow/cli/` | 74% | main (96%), visualize cmd (87%), utils (87%), evaluate cmd (83%), analyse cmd (65%), convert cmd (52%) |
|
|
394
426
|
| `dataflow/util/` | 100% | logging (100%) |
|
|
395
427
|
|
|
@@ -398,11 +430,10 @@ pytest tests/evaluate/test_evaluator.py # Single module
|
|
|
398
430
|
### 🎨 Code Quality
|
|
399
431
|
|
|
400
432
|
```bash
|
|
401
|
-
pip install -e .[dev]
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
mypy dataflow
|
|
405
|
-
flake8 dataflow tests samples # Lint
|
|
433
|
+
pip install -e .[dev] # Install dev dependencies
|
|
434
|
+
ruff check dataflow tests samples # Lint
|
|
435
|
+
ruff format --check dataflow tests samples # Format check
|
|
436
|
+
mypy dataflow # Type check
|
|
406
437
|
```
|
|
407
438
|
|
|
408
439
|
### 🔗 Pre-commit Hooks (Optional)
|
|
@@ -412,7 +443,7 @@ pip install pre-commit
|
|
|
412
443
|
pre-commit install # Install git hooks (run once)
|
|
413
444
|
|
|
414
445
|
# After this, every `git commit` auto-runs:
|
|
415
|
-
#
|
|
446
|
+
# ruff (lint, auto-fix) → ruff format → whitespace checks
|
|
416
447
|
|
|
417
448
|
pre-commit run --all-files # Manual run against all files
|
|
418
449
|
```
|
|
@@ -428,7 +459,7 @@ dataflow/
|
|
|
428
459
|
├── evaluate/ # pycocotools-based metrics, log templates
|
|
429
460
|
├── util/ # Unified logging (LogManager + format helpers)
|
|
430
461
|
└── cli/ # CLI entry point, commands, validation
|
|
431
|
-
tests/ # Unit & integration tests (
|
|
462
|
+
tests/ # Unit & integration tests (606 tests, conftest fixtures)
|
|
432
463
|
samples/ # Python API usage examples (analyse, convert, visualize, evaluate, cli)
|
|
433
464
|
assets/ # Test data (det/seg by format)
|
|
434
465
|
specs/ # Canonical specifications (evaluate/ + formats/ + modules/)
|
|
@@ -26,6 +26,7 @@ A computer vision dataset processing library — analyse, convert, visualize, an
|
|
|
26
26
|
| 🎨 **Visualize** | OpenCV rendering with color-coded classes, display & save modes | `dataflow-cv visualize yolo ...` |
|
|
27
27
|
| 📊 **Evaluate** | COCO mAP via pycocotools, single-threshold P/R/F1 per class | `dataflow-cv evaluate detection ...` |
|
|
28
28
|
| 💻 **CLI + API** | Click-based CLI with rich `--help`; Python API for pipelines | `from dataflow.convert import ...` |
|
|
29
|
+
| 🤖 **AI Skills** | Claude Code skill (`/dataflow:dataflow-cv`) for AI assistants — CLI/API reference & known gotchas | `claude plugin install dataflow@claude-skills` |
|
|
29
30
|
|
|
30
31
|
---
|
|
31
32
|
|
|
@@ -116,6 +117,12 @@ dataflow-cv convert yolo2coco --verbose images/ labels/ classes.txt output.json
|
|
|
116
117
|
dataflow-cv convert yolo2coco --no-strict images/ labels/ classes.txt output.json
|
|
117
118
|
```
|
|
118
119
|
|
|
120
|
+
> 📂 **COCO→YOLO / COCO→LabelMe output naming**: each `.txt` / `.json` is named by the
|
|
121
|
+
> image file stem — leading zeros preserved (`000001.jpg` → `000001.txt`). The
|
|
122
|
+
> converter generates `labels/` (or the `.json` files) + `classes.txt`; the
|
|
123
|
+
> `images/` directory is created but left **empty** — images are never copied,
|
|
124
|
+
> place your image files there yourself.
|
|
125
|
+
|
|
119
126
|
#### 🎨 Visualization
|
|
120
127
|
|
|
121
128
|
```bash
|
|
@@ -191,7 +198,13 @@ Two evaluation modes, distinguished by how overlap is measured:
|
|
|
191
198
|
|
|
192
199
|
```python
|
|
193
200
|
from dataflow.util.logging import LogConfig
|
|
194
|
-
from dataflow.analyse import
|
|
201
|
+
from dataflow.analyse import (
|
|
202
|
+
StatsAnalyser,
|
|
203
|
+
SplitAnalyser,
|
|
204
|
+
FilterAnalyser,
|
|
205
|
+
PartitionAnalyser,
|
|
206
|
+
SampleAnalyser,
|
|
207
|
+
)
|
|
195
208
|
from dataflow.convert import YoloAndCocoConverter
|
|
196
209
|
from dataflow.visualize import YOLOVisualizer
|
|
197
210
|
from dataflow.evaluate import DetectionEvaluator, compute_pr_f1
|
|
@@ -207,37 +220,50 @@ print(f"{result.data.total_files} images, {result.data.total_annotations} object
|
|
|
207
220
|
# Train/test split (YOLO / LabelMe)
|
|
208
221
|
splitter = SplitAnalyser(log_config=log_cfg)
|
|
209
222
|
result = splitter.analyse(
|
|
210
|
-
output_dir="output/",
|
|
211
|
-
|
|
223
|
+
output_dir="output/",
|
|
224
|
+
ratio=0.8,
|
|
225
|
+
seed=42,
|
|
226
|
+
label_dir="yolo_labels/",
|
|
227
|
+
class_file="classes.txt",
|
|
212
228
|
)
|
|
213
229
|
print(f"Train: {result.data.train_count}, Val: {result.data.val_count}")
|
|
214
230
|
|
|
215
231
|
# Split with images (both mode — labels drive, images follow by stem)
|
|
216
232
|
result = splitter.analyse(
|
|
217
|
-
output_dir="output/",
|
|
218
|
-
|
|
233
|
+
output_dir="output/",
|
|
234
|
+
ratio=0.8,
|
|
235
|
+
seed=42,
|
|
236
|
+
label_dir="yolo_labels/",
|
|
237
|
+
image_dir="images/",
|
|
219
238
|
class_file="classes.txt",
|
|
220
239
|
)
|
|
221
240
|
|
|
222
241
|
# Category filter (keep / remap categories per new classes.txt)
|
|
223
242
|
filterer = FilterAnalyser(log_config=log_cfg)
|
|
224
243
|
result = filterer.analyse(
|
|
225
|
-
"yolo_labels/",
|
|
226
|
-
|
|
244
|
+
"yolo_labels/",
|
|
245
|
+
original_class_file="classes.txt",
|
|
246
|
+
new_class_file="classes_new.txt",
|
|
247
|
+
output_dir="filtered/",
|
|
227
248
|
)
|
|
228
249
|
|
|
229
250
|
# N-way partition (YOLO / LabelMe labels; images follow by stem)
|
|
230
251
|
partitioner = PartitionAnalyser(log_config=log_cfg)
|
|
231
252
|
result = partitioner.analyse(
|
|
232
|
-
output_dir="parts/",
|
|
233
|
-
|
|
253
|
+
output_dir="parts/",
|
|
254
|
+
num=4,
|
|
255
|
+
label_dir="yolo_labels/",
|
|
256
|
+
image_dir="images/",
|
|
234
257
|
)
|
|
235
258
|
|
|
236
259
|
# File sampling (labels, images, or both — random or sequential)
|
|
237
260
|
sampler = SampleAnalyser(log_config=log_cfg)
|
|
238
261
|
result = sampler.analyse(
|
|
239
|
-
output_dir="sampled/",
|
|
240
|
-
|
|
262
|
+
output_dir="sampled/",
|
|
263
|
+
count=10,
|
|
264
|
+
label_dir="yolo_labels/",
|
|
265
|
+
shuffle=True,
|
|
266
|
+
seed=42,
|
|
241
267
|
)
|
|
242
268
|
|
|
243
269
|
# ── Convert ──────────────────────────────────────────
|
|
@@ -245,22 +271,30 @@ result = sampler.analyse(
|
|
|
245
271
|
log_cfg = LogConfig(name="convert", verbose=True)
|
|
246
272
|
converter = YoloAndCocoConverter(source_to_target=True, log_config=log_cfg, strict_mode=True)
|
|
247
273
|
result = converter.convert(
|
|
248
|
-
source_path="yolo_labels/",
|
|
249
|
-
|
|
274
|
+
source_path="yolo_labels/",
|
|
275
|
+
target_path="anno.json",
|
|
276
|
+
class_file="classes.txt",
|
|
277
|
+
image_dir="images/",
|
|
250
278
|
)
|
|
251
279
|
|
|
252
280
|
# YOLO predictions → COCO (prediction mode)
|
|
253
281
|
converter = YoloAndCocoConverter(source_to_target=True, prediction=True)
|
|
254
282
|
result = converter.convert(
|
|
255
|
-
source_path="yolo_preds/",
|
|
256
|
-
|
|
283
|
+
source_path="yolo_preds/",
|
|
284
|
+
target_path="pred.json",
|
|
285
|
+
class_file="classes.txt",
|
|
286
|
+
image_dir="images/",
|
|
257
287
|
)
|
|
258
288
|
|
|
259
289
|
# ── Visualize ────────────────────────────────────────
|
|
260
290
|
visualizer = YOLOVisualizer(
|
|
261
|
-
label_dir="yolo_labels/",
|
|
262
|
-
|
|
263
|
-
|
|
291
|
+
label_dir="yolo_labels/",
|
|
292
|
+
image_dir="images/",
|
|
293
|
+
class_file="classes.txt",
|
|
294
|
+
is_show=True,
|
|
295
|
+
is_save=True,
|
|
296
|
+
output_dir="visualized/",
|
|
297
|
+
log_config=log_cfg,
|
|
264
298
|
)
|
|
265
299
|
result = visualizer.visualize()
|
|
266
300
|
|
|
@@ -330,7 +364,7 @@ For detailed developer guidance including advanced test commands, debugging, and
|
|
|
330
364
|
|
|
331
365
|
### 🧪 Testing
|
|
332
366
|
|
|
333
|
-
**
|
|
367
|
+
**606 tests, 80% code coverage (5532 statements).**
|
|
334
368
|
|
|
335
369
|
```bash
|
|
336
370
|
pytest # All tests
|
|
@@ -344,11 +378,11 @@ pytest tests/evaluate/test_evaluator.py # Single module
|
|
|
344
378
|
|
|
345
379
|
| Module | Coverage | Highlights |
|
|
346
380
|
|--------|:--------:|------------|
|
|
347
|
-
| `dataflow/label/` | 71% | models (84%), base (82%), utils (
|
|
348
|
-
| `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (
|
|
349
|
-
| `dataflow/convert/` |
|
|
381
|
+
| `dataflow/label/` | 71% | models (84%), base (82%), utils (83%), coco_handler (74%), labelme_handler (71%), yolo_handler (61%) |
|
|
382
|
+
| `dataflow/analyse/` | 84% | base (99%), log_templates (92%), sample (87%), utils (84%), split (85%), stats (82%), filter (76%), partition (74%) |
|
|
383
|
+
| `dataflow/convert/` | 86% | labelme_and_yolo (93%), yolo_and_coco (89%), utils (88%), coco_and_labelme (88%), log_templates (85%), base (81%), rle (80%) |
|
|
350
384
|
| `dataflow/visualize/` | 81% | yolo_vis (100%), labelme_vis (100%), coco_vis (93%), base (76%) |
|
|
351
|
-
| `dataflow/evaluate/` |
|
|
385
|
+
| `dataflow/evaluate/` | 90% | evaluator (100%), result (99%), metrics (93%), base (90%), utils (87%) |
|
|
352
386
|
| `dataflow/cli/` | 74% | main (96%), visualize cmd (87%), utils (87%), evaluate cmd (83%), analyse cmd (65%), convert cmd (52%) |
|
|
353
387
|
| `dataflow/util/` | 100% | logging (100%) |
|
|
354
388
|
|
|
@@ -357,11 +391,10 @@ pytest tests/evaluate/test_evaluator.py # Single module
|
|
|
357
391
|
### 🎨 Code Quality
|
|
358
392
|
|
|
359
393
|
```bash
|
|
360
|
-
pip install -e .[dev]
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
mypy dataflow
|
|
364
|
-
flake8 dataflow tests samples # Lint
|
|
394
|
+
pip install -e .[dev] # Install dev dependencies
|
|
395
|
+
ruff check dataflow tests samples # Lint
|
|
396
|
+
ruff format --check dataflow tests samples # Format check
|
|
397
|
+
mypy dataflow # Type check
|
|
365
398
|
```
|
|
366
399
|
|
|
367
400
|
### 🔗 Pre-commit Hooks (Optional)
|
|
@@ -371,7 +404,7 @@ pip install pre-commit
|
|
|
371
404
|
pre-commit install # Install git hooks (run once)
|
|
372
405
|
|
|
373
406
|
# After this, every `git commit` auto-runs:
|
|
374
|
-
#
|
|
407
|
+
# ruff (lint, auto-fix) → ruff format → whitespace checks
|
|
375
408
|
|
|
376
409
|
pre-commit run --all-files # Manual run against all files
|
|
377
410
|
```
|
|
@@ -387,7 +420,7 @@ dataflow/
|
|
|
387
420
|
├── evaluate/ # pycocotools-based metrics, log templates
|
|
388
421
|
├── util/ # Unified logging (LogManager + format helpers)
|
|
389
422
|
└── cli/ # CLI entry point, commands, validation
|
|
390
|
-
tests/ # Unit & integration tests (
|
|
423
|
+
tests/ # Unit & integration tests (606 tests, conftest fixtures)
|
|
391
424
|
samples/ # Python API usage examples (analyse, convert, visualize, evaluate, cli)
|
|
392
425
|
assets/ # Test data (det/seg by format)
|
|
393
426
|
specs/ # Canonical specifications (evaluate/ + formats/ + modules/)
|
|
@@ -24,8 +24,7 @@ from .log_templates import (
|
|
|
24
24
|
format_filter_result,
|
|
25
25
|
)
|
|
26
26
|
from .utils import create_handler, detect_format, load_class_names
|
|
27
|
-
from dataflow.label.models import
|
|
28
|
-
ObjectAnnotation)
|
|
27
|
+
from dataflow.label.models import DatasetAnnotations, ImageAnnotation, ObjectAnnotation
|
|
29
28
|
|
|
30
29
|
|
|
31
30
|
class FilterAnalyser(BaseAnalyser):
|
|
@@ -99,17 +98,14 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
99
98
|
for new_id, name in new_classes.items():
|
|
100
99
|
if name in name_to_old_id:
|
|
101
100
|
old_id = name_to_old_id[name]
|
|
102
|
-
mapping = CategoryMapping(
|
|
103
|
-
new_id=new_id, old_id=old_id, name=name
|
|
104
|
-
)
|
|
101
|
+
mapping = CategoryMapping(new_id=new_id, old_id=old_id, name=name)
|
|
105
102
|
old_to_new[old_id] = mapping
|
|
106
103
|
kept.append(mapping)
|
|
107
104
|
else:
|
|
108
105
|
missing.append(name)
|
|
109
106
|
if logger:
|
|
110
107
|
logger.warning(
|
|
111
|
-
f'Category "{name}" in new class file not '
|
|
112
|
-
f"found in source — skipping"
|
|
108
|
+
f'Category "{name}" in new class file not found in source — skipping'
|
|
113
109
|
)
|
|
114
110
|
|
|
115
111
|
# Build removed list
|
|
@@ -134,10 +130,7 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
134
130
|
total_after = 0
|
|
135
131
|
|
|
136
132
|
for image_ann in dataset.images:
|
|
137
|
-
filtered = [
|
|
138
|
-
obj for obj in image_ann.objects
|
|
139
|
-
if obj.class_id in old_to_new
|
|
140
|
-
]
|
|
133
|
+
filtered = [obj for obj in image_ann.objects if obj.class_id in old_to_new]
|
|
141
134
|
for obj in filtered:
|
|
142
135
|
mapping = old_to_new[obj.class_id]
|
|
143
136
|
obj.class_id = mapping.new_id
|
|
@@ -202,9 +195,7 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
202
195
|
return result
|
|
203
196
|
|
|
204
197
|
if not new_classes:
|
|
205
|
-
result.add_error(
|
|
206
|
-
f"No valid class names in new class file: {new_class_file}"
|
|
207
|
-
)
|
|
198
|
+
result.add_error(f"No valid class names in new class file: {new_class_file}")
|
|
208
199
|
return result
|
|
209
200
|
|
|
210
201
|
# ---- 3. Detect format + create handler ------------------------
|
|
@@ -259,9 +250,7 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
259
250
|
)
|
|
260
251
|
|
|
261
252
|
if not old_to_new:
|
|
262
|
-
result.add_error(
|
|
263
|
-
"No matching categories between source and new class file"
|
|
264
|
-
)
|
|
253
|
+
result.add_error("No matching categories between source and new class file")
|
|
265
254
|
return result
|
|
266
255
|
|
|
267
256
|
# ---- 5. Ensure output directory --------------------------------
|
|
@@ -304,14 +293,16 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
304
293
|
# rather than mutating the original (avoid
|
|
305
294
|
# aliasing — the original may be reused by the
|
|
306
295
|
# iterator or shared across images).
|
|
307
|
-
filtered_objects.append(
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
296
|
+
filtered_objects.append(
|
|
297
|
+
ObjectAnnotation(
|
|
298
|
+
class_id=mapping.new_id,
|
|
299
|
+
class_name=mapping.name,
|
|
300
|
+
bbox=obj.bbox,
|
|
301
|
+
segmentation=obj.segmentation,
|
|
302
|
+
confidence=obj.confidence,
|
|
303
|
+
is_crowd=obj.is_crowd,
|
|
304
|
+
)
|
|
305
|
+
)
|
|
315
306
|
total_after += len(filtered_objects)
|
|
316
307
|
|
|
317
308
|
if filtered_objects:
|
|
@@ -328,9 +319,7 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
328
319
|
wr = write_handler.write_one(filtered_img, output_dir)
|
|
329
320
|
if not wr.success:
|
|
330
321
|
for err in wr.errors:
|
|
331
|
-
result.add_error(
|
|
332
|
-
f"Write {image_ann.image_id}: {err}"
|
|
333
|
-
)
|
|
322
|
+
result.add_error(f"Write {image_ann.image_id}: {err}")
|
|
334
323
|
return result
|
|
335
324
|
except Exception as e:
|
|
336
325
|
result.add_error(f"Failed during streaming filter: {e}")
|
|
@@ -342,23 +331,19 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
342
331
|
total_files = dataset.num_images
|
|
343
332
|
total_before = dataset.num_objects
|
|
344
333
|
|
|
345
|
-
total_files_with_annotations, total_after = (
|
|
346
|
-
|
|
334
|
+
total_files_with_annotations, total_after = self._filter_dataset_images(
|
|
335
|
+
dataset, old_to_new
|
|
347
336
|
)
|
|
348
337
|
|
|
349
338
|
# Update categories
|
|
350
|
-
dataset.categories = {
|
|
351
|
-
km.new_id: km.name for km in kept_categories
|
|
352
|
-
}
|
|
339
|
+
dataset.categories = {km.new_id: km.name for km in kept_categories}
|
|
353
340
|
|
|
354
341
|
try:
|
|
355
342
|
for image_ann in dataset.images:
|
|
356
343
|
wr = handler.write_one(image_ann, output_dir)
|
|
357
344
|
if not wr.success:
|
|
358
345
|
for err in wr.errors:
|
|
359
|
-
result.add_error(
|
|
360
|
-
f"Write {image_ann.image_id}: {err}"
|
|
361
|
-
)
|
|
346
|
+
result.add_error(f"Write {image_ann.image_id}: {err}")
|
|
362
347
|
return result
|
|
363
348
|
except Exception as e:
|
|
364
349
|
result.add_error(f"Failed to write filtered output: {e}")
|
|
@@ -369,14 +354,12 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
369
354
|
total_files = dataset.num_images
|
|
370
355
|
total_before = dataset.num_objects
|
|
371
356
|
|
|
372
|
-
total_files_with_annotations, total_after = (
|
|
373
|
-
|
|
357
|
+
total_files_with_annotations, total_after = self._filter_dataset_images(
|
|
358
|
+
dataset, old_to_new
|
|
374
359
|
)
|
|
375
360
|
|
|
376
361
|
# Update categories to match new class file
|
|
377
|
-
dataset.categories = {
|
|
378
|
-
km.new_id: km.name for km in kept_categories
|
|
379
|
-
}
|
|
362
|
+
dataset.categories = {km.new_id: km.name for km in kept_categories}
|
|
380
363
|
|
|
381
364
|
output_path = output_dir / label_path.name
|
|
382
365
|
try:
|
|
@@ -413,29 +396,20 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
413
396
|
if missing_categories:
|
|
414
397
|
s = "y" if len(missing_categories) == 1 else "ies"
|
|
415
398
|
result.add_warning(
|
|
416
|
-
f"{len(missing_categories)} categor{s} in new class "
|
|
417
|
-
f"file not found in source"
|
|
399
|
+
f"{len(missing_categories)} categor{s} in new class file not found in source"
|
|
418
400
|
)
|
|
419
401
|
|
|
420
402
|
if total_after == 0 and total_before > 0:
|
|
421
|
-
result.add_warning(
|
|
422
|
-
"All annotations were filtered out — output files are empty"
|
|
423
|
-
)
|
|
403
|
+
result.add_warning("All annotations were filtered out — output files are empty")
|
|
424
404
|
|
|
425
405
|
# ---- 9. Log output ---------------------------------------------
|
|
426
406
|
self._log_info(
|
|
427
|
-
format_analyse_header(
|
|
428
|
-
"Category Filter", label_path, f"{fmt} (auto-detected)"
|
|
429
|
-
)
|
|
407
|
+
format_analyse_header("Category Filter", label_path, f"{fmt} (auto-detected)")
|
|
430
408
|
)
|
|
431
409
|
self._log_info(
|
|
432
|
-
f" Original class: {original_class_file.name} "
|
|
433
|
-
f"({len(original_classes)} categories)"
|
|
434
|
-
)
|
|
435
|
-
self._log_info(
|
|
436
|
-
f" New class: {new_class_file.name} "
|
|
437
|
-
f"({len(new_classes)} categories)\n"
|
|
410
|
+
f" Original class: {original_class_file.name} ({len(original_classes)} categories)"
|
|
438
411
|
)
|
|
412
|
+
self._log_info(f" New class: {new_class_file.name} ({len(new_classes)} categories)\n")
|
|
439
413
|
self._log_info(
|
|
440
414
|
format_filter_result(
|
|
441
415
|
total_files=total_files,
|
|
@@ -449,8 +423,6 @@ class FilterAnalyser(BaseAnalyser):
|
|
|
449
423
|
)
|
|
450
424
|
)
|
|
451
425
|
if result.log_path:
|
|
452
|
-
self._log_info(
|
|
453
|
-
format_analyse_result("✓ Success", result.log_path)
|
|
454
|
-
)
|
|
426
|
+
self._log_info(format_analyse_result("✓ Success", result.log_path))
|
|
455
427
|
|
|
456
428
|
return result
|