airo-dataset-tools 2025.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- airo_dataset_tools-2025.4.0/PKG-INFO +24 -0
- airo_dataset_tools-2025.4.0/README.md +20 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/__init__.py +5 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/cli.py +136 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/__init__.py +0 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/change_coco_images_prefix.py +40 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/coco_instances_to_yolo.py +143 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/fiftyone_viewer.py +29 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/merge_datasets.py +132 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/split_dataset.py +112 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/transform_dataset.py +218 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/transforms.py +37 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/cvat_labeling/__init__.py +0 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/cvat_labeling/convert_cvat_to_coco.py +302 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/cvat_labeling/load_xml_to_dict.py +24 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/data_parsers/__init__.py +4 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/data_parsers/camera_intrinsics.py +70 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/data_parsers/coco.py +205 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/data_parsers/cvat_images.py +142 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/data_parsers/pose.py +58 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/py.typed +0 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools/segmentation_mask_converter.py +186 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/PKG-INFO +24 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/SOURCES.txt +38 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/dependency_links.txt +1 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/entry_points.txt +2 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/requires.txt +14 -0
- airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/top_level.txt +1 -0
- airo_dataset_tools-2025.4.0/setup.cfg +4 -0
- airo_dataset_tools-2025.4.0/setup.py +35 -0
- airo_dataset_tools-2025.4.0/test/test_camera_intrinsics.py +54 -0
- airo_dataset_tools-2025.4.0/test/test_coco_create.py +181 -0
- airo_dataset_tools-2025.4.0/test/test_coco_load.py +101 -0
- airo_dataset_tools-2025.4.0/test/test_coco_merge.py +29 -0
- airo_dataset_tools-2025.4.0/test/test_coco_split.py +33 -0
- airo_dataset_tools-2025.4.0/test/test_cvat_images_load.py +18 -0
- airo_dataset_tools-2025.4.0/test/test_cvat_to_coco_conversion.py +8 -0
- airo_dataset_tools-2025.4.0/test/test_pillow_resize.py +9 -0
- airo_dataset_tools-2025.4.0/test/test_pose.py +48 -0
- airo_dataset_tools-2025.4.0/test/test_segmentation_mask.py +62 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: airo-dataset-tools
|
|
3
|
+
Version: 2025.4.0
|
|
4
|
+
Summary: Scripts for loading and converting datasets for the Ghent University AI and Robotics Lab
|
|
5
|
+
Author: Victor-Louis De Gusseme
|
|
6
|
+
Author-email: victorlouisdg@gmail.com
|
|
7
|
+
Requires-Dist: numpy<2.0
|
|
8
|
+
Requires-Dist: pydantic>2.0.0
|
|
9
|
+
Requires-Dist: opencv-contrib-python==4.8.1.78
|
|
10
|
+
Requires-Dist: opencv-python-headless==4.8.1.78
|
|
11
|
+
Requires-Dist: pycocotools
|
|
12
|
+
Requires-Dist: xmltodict
|
|
13
|
+
Requires-Dist: tqdm
|
|
14
|
+
Requires-Dist: fiftyone
|
|
15
|
+
Requires-Dist: Pillow
|
|
16
|
+
Requires-Dist: types-Pillow
|
|
17
|
+
Requires-Dist: albumentations
|
|
18
|
+
Requires-Dist: click
|
|
19
|
+
Requires-Dist: airo-typing==2025.4.0
|
|
20
|
+
Requires-Dist: airo-spatial-algebra==2025.4.0
|
|
21
|
+
Dynamic: author
|
|
22
|
+
Dynamic: author-email
|
|
23
|
+
Dynamic: requires-dist
|
|
24
|
+
Dynamic: summary
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# airo-dataset-tools
|
|
2
|
+
Tools for working with datasets.
|
|
3
|
+
They fall into two categories:
|
|
4
|
+
|
|
5
|
+
[**COCO related tools**](airo_dataset_tools/coco_tools/README.md):
|
|
6
|
+
* COCO dataset loading (and creation)
|
|
7
|
+
* FiftyOne visualisation
|
|
8
|
+
* Albumentation transforms
|
|
9
|
+
* COCO to YOLO conversion.
|
|
10
|
+
* CVAT labeling workflow
|
|
11
|
+
|
|
12
|
+
[**Data formats**](airo_dataset_tools/data_parsers/README.md):
|
|
13
|
+
* 3D poses
|
|
14
|
+
* Camera instrinsics
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
> [Pydantic](https://docs.pydantic.dev/latest/) is used heavily throughout this package.
|
|
18
|
+
It allows you to easily create Python objects that can be saved and loaded to and from JSON files.
|
|
19
|
+
|
|
20
|
+
[**CVAT labeling workflow & tools**](airo_dataset_tools/cvat_labeling/readme.md)
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
# init files are still required in modern python!
|
|
2
|
+
# https://peps.python.org/pep-0420/ introduced implicit namespace packages
|
|
3
|
+
# but for building and many toolings, you still need to have __init__ files (at least in the root of the package).
|
|
4
|
+
# e.g. if you remove this init file and try to build with pip install .
|
|
5
|
+
# you won't be able to import the dummy module.
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
"""CLI interface for this package"""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
from typing import List, Optional
|
|
6
|
+
|
|
7
|
+
import click
|
|
8
|
+
from airo_dataset_tools.coco_tools.change_coco_images_prefix import change_coco_json_image_prefix
|
|
9
|
+
from airo_dataset_tools.coco_tools.coco_instances_to_yolo import create_yolo_dataset_from_coco_instances_dataset
|
|
10
|
+
from airo_dataset_tools.coco_tools.fiftyone_viewer import view_coco_dataset
|
|
11
|
+
from airo_dataset_tools.coco_tools.merge_datasets import merge_coco_datasets
|
|
12
|
+
from airo_dataset_tools.coco_tools.split_dataset import split_and_save_coco_dataset
|
|
13
|
+
from airo_dataset_tools.coco_tools.transform_dataset import resize_coco_dataset
|
|
14
|
+
from airo_dataset_tools.cvat_labeling.convert_cvat_to_coco import cvat_image_to_coco
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@click.group()
|
|
18
|
+
def cli() -> None:
|
|
19
|
+
"""CLI entrypoint for airo-dataset-tools"""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@cli.command(name="fiftyone-coco-viewer") # no help, takes the docstring of the function.
|
|
23
|
+
@click.argument("annotations-json-path", type=click.Path(exists=True))
|
|
24
|
+
@click.option(
|
|
25
|
+
"--dataset-dir",
|
|
26
|
+
required=False,
|
|
27
|
+
type=click.Path(exists=True),
|
|
28
|
+
help="optional directory relative to which the image paths in the coco dataset are specified",
|
|
29
|
+
)
|
|
30
|
+
@click.option(
|
|
31
|
+
"--label-types",
|
|
32
|
+
"-l",
|
|
33
|
+
multiple=True,
|
|
34
|
+
type=click.Choice(["detections", "segmentations", "keypoints"]),
|
|
35
|
+
help="add an argument for each label type you want to load (default: all)",
|
|
36
|
+
)
|
|
37
|
+
def view_coco_dataset_cli(
|
|
38
|
+
annotations_json_path: str, dataset_dir: str, label_types: Optional[List[str]] = None
|
|
39
|
+
) -> None:
|
|
40
|
+
"""Explore COCO dataset with FiftyOne"""
|
|
41
|
+
view_coco_dataset(annotations_json_path, dataset_dir, label_types)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@cli.command(name="convert-cvat-to-coco-keypoints")
|
|
45
|
+
@click.argument("cvat_xml_file", type=str, required=True)
|
|
46
|
+
@click.argument("coco-categories-json-file", type=str, required=True)
|
|
47
|
+
@click.option("--add-bbox", is_flag=True, default=False, help="include bounding box in coco annotations")
|
|
48
|
+
@click.option("--add-segmentation", is_flag=True, default=False, help="include segmentation in coco annotations")
|
|
49
|
+
def convert_cvat_to_coco_cli(
|
|
50
|
+
cvat_xml_file: str, coco_categories_json_file: str, add_bbox: bool, add_segmentation: bool
|
|
51
|
+
) -> None:
|
|
52
|
+
"""Convert CVAT XML to COCO keypoints json according to specified coco categories"""
|
|
53
|
+
coco = cvat_image_to_coco(
|
|
54
|
+
cvat_xml_file, coco_categories_json_file, add_bbox=add_bbox, add_segmentation=add_segmentation
|
|
55
|
+
)
|
|
56
|
+
path = os.path.dirname(cvat_xml_file)
|
|
57
|
+
filename = os.path.basename(cvat_xml_file)
|
|
58
|
+
path = os.path.join(path, filename.split(".")[0] + ".json")
|
|
59
|
+
with open(path, "w") as file:
|
|
60
|
+
json.dump(coco, file)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@cli.command(name="resize-coco-dataset")
|
|
64
|
+
@click.argument("annotations-json-path", type=click.Path(exists=True))
|
|
65
|
+
@click.option("--width", type=int, required=True)
|
|
66
|
+
@click.option("--height", type=int, required=True)
|
|
67
|
+
@click.option("--target-dataset-dir", type=str, required=False)
|
|
68
|
+
def resize_coco_dataset_cli(
|
|
69
|
+
annotations_json_path: str, width: int, height: int, target_dataset_dir: Optional[str]
|
|
70
|
+
) -> None:
|
|
71
|
+
"""Resize a COCO dataset. Will create a new directory with the resized dataset at the specified target_dataset_dir.
|
|
72
|
+
Dataset is assumed to be
|
|
73
|
+
/dir
|
|
74
|
+
annotations.json # contains relative paths w.r.t. /dir
|
|
75
|
+
...
|
|
76
|
+
"""
|
|
77
|
+
resize_coco_dataset(annotations_json_path, width, height, target_dataset_dir=target_dataset_dir)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@cli.command(name="coco-instances-to-yolo")
|
|
81
|
+
@click.option("--coco-json", type=str)
|
|
82
|
+
@click.option("--target-dir", type=str)
|
|
83
|
+
@click.option("--use-segmentation", is_flag=True)
|
|
84
|
+
def coco_intances_to_yolo(coco_json: str, target_dir: str, use_segmentation: bool) -> None:
|
|
85
|
+
"""Create a YOLO detections/segmentations dataset from a coco instances dataset"""
|
|
86
|
+
print(f"converting coco instances dataset {coco_json} to yolo dataset {target_dir}")
|
|
87
|
+
create_yolo_dataset_from_coco_instances_dataset(coco_json, target_dir, use_segmentation=use_segmentation)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@cli.command(name="split-coco-dataset")
|
|
91
|
+
@click.argument("json-path", type=click.Path(exists=True))
|
|
92
|
+
@click.option("--split-ratio", type=float, multiple=True, required=True)
|
|
93
|
+
@click.option("--shuffle-before-splitting", is_flag=True, default=True)
|
|
94
|
+
def split_coco_dataset_cli(json_path: str, split_ratio: List[float], shuffle_before_splitting: bool) -> None:
|
|
95
|
+
"""Split a COCO dataset into subsets according to the specified relative ratios and save them to disk.
|
|
96
|
+
Images are split with their corresponding annotations. No guarantees on class balance or annotation ratios.
|
|
97
|
+
|
|
98
|
+
If two ratios are specified, the dataset will be split into two subsets. these will be called train/val by default.
|
|
99
|
+
If three ratios are specified, the dataset will be split into three subsets. these will be called train/val/test by default.
|
|
100
|
+
|
|
101
|
+
e.g. split-coco-dataset <path> --split-ratio 0.8 --split-ratio 0.2
|
|
102
|
+
"""
|
|
103
|
+
split_and_save_coco_dataset(json_path, split_ratio, shuffle_before_splitting=shuffle_before_splitting)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@cli.command(name="change-coco-images-prefix")
|
|
107
|
+
@click.argument("coco-json", type=click.Path(exists=True))
|
|
108
|
+
@click.option("--current-prefix", type=str, required=True)
|
|
109
|
+
@click.option("--new-prefix", type=str, required=True)
|
|
110
|
+
@click.option("--target-json-path", type=str)
|
|
111
|
+
def change_coco_images_prefix_cli(
|
|
112
|
+
coco_json: str, current_prefix: str, new_prefix: str, target_json_path: Optional[str] = None
|
|
113
|
+
) -> None:
|
|
114
|
+
"""change the prefix of images in a coco dataset."""
|
|
115
|
+
return change_coco_json_image_prefix(coco_json, current_prefix, new_prefix, target_json_path)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
@cli.command(name="merge-coco-datasets")
|
|
119
|
+
@click.argument("coco-json-1", type=click.Path(exists=True))
|
|
120
|
+
@click.argument("coco-json-2", type=click.Path(exists=True))
|
|
121
|
+
@click.option(
|
|
122
|
+
"--target-json-path",
|
|
123
|
+
type=str,
|
|
124
|
+
help="optional path to save the merged dataset to. If none is provided, a new directory will be created in the parent directory of coco-json-1 ",
|
|
125
|
+
)
|
|
126
|
+
def merge_coco_datasets_cli(coco_json_1: str, coco_json_2: str, target_json_path: Optional[str] = None) -> None:
|
|
127
|
+
"""merge two coco datasets into a single dataset."""
|
|
128
|
+
if not target_json_path:
|
|
129
|
+
target_json_path = os.path.join(os.path.dirname(coco_json_1), "merged")
|
|
130
|
+
os.makedirs(target_json_path, exist_ok=True)
|
|
131
|
+
target_json_path = os.path.join(target_json_path, "annotations.json")
|
|
132
|
+
return merge_coco_datasets(coco_json_1, coco_json_2, target_json_path)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
if __name__ == "__main__":
|
|
136
|
+
cli()
|
|
File without changes
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from typing import Optional
|
|
3
|
+
|
|
4
|
+
from airo_dataset_tools.data_parsers.coco import CocoInstancesDataset
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def change_coco_dataset_image_prefix(
|
|
8
|
+
coco_dataset: CocoInstancesDataset, current_prefix: str, target_prefix: str
|
|
9
|
+
) -> CocoInstancesDataset:
|
|
10
|
+
"""Change the prefix of the image paths in a COCO dataset. Can be used to change the base directory of the images in a dataset.
|
|
11
|
+
|
|
12
|
+
Args:
|
|
13
|
+
coco_dataset: the dataset to modify
|
|
14
|
+
current_prefix: the current prefix of the image paths in the dataset
|
|
15
|
+
target_prefix: the target prefix of the image paths in the dataset
|
|
16
|
+
|
|
17
|
+
e.g.
|
|
18
|
+
|
|
19
|
+
if the images are currently relative to the image folder inside the dataset and you want to make them relative to the dataset folder:
|
|
20
|
+
change_coco_image_base_dir(coco_dataset, "", "images/")
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
for image in coco_dataset.images:
|
|
24
|
+
image.file_name = image.file_name.removeprefix(current_prefix)
|
|
25
|
+
image.file_name = target_prefix + image.file_name
|
|
26
|
+
|
|
27
|
+
return coco_dataset
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def change_coco_json_image_prefix(
|
|
31
|
+
coco_json_file: str, current_prefix: str, target_prefix: str, target_json_file: Optional[str]
|
|
32
|
+
) -> None:
|
|
33
|
+
with open(coco_json_file, "r") as f:
|
|
34
|
+
coco_dataset = CocoInstancesDataset(**json.load(f))
|
|
35
|
+
|
|
36
|
+
coco_dataset = change_coco_dataset_image_prefix(coco_dataset, current_prefix, target_prefix)
|
|
37
|
+
if target_json_file is None:
|
|
38
|
+
target_json_file = coco_json_file
|
|
39
|
+
with open(target_json_file, "w") as f:
|
|
40
|
+
json.dump(coco_dataset.model_dump(), f)
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import pathlib
|
|
3
|
+
from collections import defaultdict
|
|
4
|
+
|
|
5
|
+
import cv2
|
|
6
|
+
import tqdm
|
|
7
|
+
from airo_dataset_tools.data_parsers.coco import CocoInstancesDataset
|
|
8
|
+
from airo_dataset_tools.segmentation_mask_converter import BinarySegmentationMask
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def create_yolo_dataset_from_coco_instances_dataset(
|
|
12
|
+
coco_dataset_json_path: str, target_directory: str, use_segmentation: bool = False
|
|
13
|
+
) -> None:
|
|
14
|
+
"""Converts a coco dataset to a yolo dataset, either segmentations or detections
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
coco dataset is expected:
|
|
18
|
+
dataset/
|
|
19
|
+
images/
|
|
20
|
+
<>.json # containing paths relative to /dataset
|
|
21
|
+
|
|
22
|
+
yolo dataset format will be
|
|
23
|
+
|
|
24
|
+
dataset/
|
|
25
|
+
images/
|
|
26
|
+
labels/
|
|
27
|
+
<>.txt # paths as in coco dataset /images subdir.
|
|
28
|
+
|
|
29
|
+
# ultralytics uses an additional <>.json, this has to be created manually..
|
|
30
|
+
# we do export a .names file, as in the darknet yolo format.
|
|
31
|
+
# each line contains a class name, the line number is the class id
|
|
32
|
+
obj.names
|
|
33
|
+
Args:
|
|
34
|
+
coco_dataset_json_path: _description_
|
|
35
|
+
target_directory: _description_
|
|
36
|
+
use_segmentation: create a segmentation dataset instead of a detection dataset (requires segmentation annotations)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
quiz question: how many people would have written a similar converter in the past?...
|
|
40
|
+
|
|
41
|
+
"""
|
|
42
|
+
_coco_dataset_json_path = pathlib.Path(coco_dataset_json_path)
|
|
43
|
+
_target_directory = pathlib.Path(target_directory)
|
|
44
|
+
|
|
45
|
+
# load the json file into the coco dataset object
|
|
46
|
+
annotations = json.load(open(_coco_dataset_json_path, "r"))
|
|
47
|
+
coco_dataset = CocoInstancesDataset(**annotations)
|
|
48
|
+
|
|
49
|
+
target_dir = pathlib.Path(_target_directory)
|
|
50
|
+
target_dir.mkdir(parents=True, exist_ok=True)
|
|
51
|
+
image_dir = target_dir / "images"
|
|
52
|
+
label_dir = target_dir / "labels"
|
|
53
|
+
image_dir.mkdir(parents=True, exist_ok=True)
|
|
54
|
+
label_dir.mkdir(parents=True, exist_ok=True)
|
|
55
|
+
|
|
56
|
+
# create the yolo dataset
|
|
57
|
+
annotations = coco_dataset.annotations
|
|
58
|
+
image_id_to_image = {image.id: image for image in coco_dataset.images}
|
|
59
|
+
image_id_to_annotations = defaultdict(list)
|
|
60
|
+
for annotation in annotations:
|
|
61
|
+
image_id_to_annotations[annotation.image_id].append(annotation)
|
|
62
|
+
|
|
63
|
+
category_id_to_category = {category.id: category for category in coco_dataset.categories}
|
|
64
|
+
# sort by original COCO ID, but whereas COCO does not care about ID ranges, YOLO expects them to be in the range 0..N-1
|
|
65
|
+
# so we sort them by ID and then use the index as the new ID
|
|
66
|
+
yolo_category_index = list(sorted(category_id_to_category.values(), key=lambda category: category.id))
|
|
67
|
+
|
|
68
|
+
for image_id, annotations in tqdm.tqdm(image_id_to_annotations.items()):
|
|
69
|
+
coco_image = image_id_to_image[image_id]
|
|
70
|
+
image_path = pathlib.Path(coco_image.file_name)
|
|
71
|
+
if not image_path.is_absolute():
|
|
72
|
+
image_path = _coco_dataset_json_path.parent / image_path
|
|
73
|
+
|
|
74
|
+
relative_image_path = image_path.relative_to(_coco_dataset_json_path.parent)
|
|
75
|
+
|
|
76
|
+
# ultralytics parser finds the 'latest' occurance of 'images' in the dataset path,
|
|
77
|
+
# so we need to replace any occurance of 'image' with 'img'
|
|
78
|
+
# https://github.com/ultralytics/ultralytics/issues/3581
|
|
79
|
+
|
|
80
|
+
relative_image_path_str = str(relative_image_path).replace("image", "img")
|
|
81
|
+
relative_image_path = pathlib.Path(relative_image_path_str)
|
|
82
|
+
|
|
83
|
+
image = cv2.imread(str(image_path))
|
|
84
|
+
height, width, _ = image.shape
|
|
85
|
+
|
|
86
|
+
label_path = label_dir / f"{relative_image_path.with_suffix('')}.txt"
|
|
87
|
+
label_path.parent.mkdir(parents=True, exist_ok=True)
|
|
88
|
+
|
|
89
|
+
with open(label_path, "w") as file:
|
|
90
|
+
for annotation in annotations:
|
|
91
|
+
category = category_id_to_category[annotation.category_id]
|
|
92
|
+
yolo_id = yolo_category_index.index(category)
|
|
93
|
+
if use_segmentation:
|
|
94
|
+
segmentation = annotation.segmentation
|
|
95
|
+
# convert to **single** polygon
|
|
96
|
+
segmentation = BinarySegmentationMask.from_coco_segmentation_mask(segmentation, width, height)
|
|
97
|
+
segmentation = segmentation.as_single_polygon
|
|
98
|
+
|
|
99
|
+
if segmentation is None:
|
|
100
|
+
# should actually never happen as each annotation is assumed to have a segmentation if you pass use_segmentation=True
|
|
101
|
+
# but we filter it for convenience to deal with edge cases
|
|
102
|
+
print(f"skipping annotation for image {image_path}, as it has no segmentation")
|
|
103
|
+
continue
|
|
104
|
+
|
|
105
|
+
file.write(f"{yolo_id}")
|
|
106
|
+
for (x, y) in zip(segmentation[0::2], segmentation[1::2]):
|
|
107
|
+
file.write(f" {x/width} {y/height}")
|
|
108
|
+
file.write("\n")
|
|
109
|
+
|
|
110
|
+
else:
|
|
111
|
+
x, y, w, h = annotation.bbox
|
|
112
|
+
x_center = x + w / 2
|
|
113
|
+
y_center = y + h / 2
|
|
114
|
+
x_center /= width
|
|
115
|
+
y_center /= height
|
|
116
|
+
w /= width
|
|
117
|
+
h /= height
|
|
118
|
+
file.write(f"{yolo_id} {x_center} {y_center} {w} {h}\n")
|
|
119
|
+
|
|
120
|
+
image_target_path = image_dir / relative_image_path
|
|
121
|
+
image_target_path.parent.mkdir(parents=True, exist_ok=True)
|
|
122
|
+
cv2.imwrite(str(image_target_path), image)
|
|
123
|
+
|
|
124
|
+
# create the obj.names file
|
|
125
|
+
with open(target_dir / "obj.names", "w") as file:
|
|
126
|
+
# sort categories by id
|
|
127
|
+
for category in yolo_category_index:
|
|
128
|
+
file.write(f"{category.name}\n")
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
if __name__ == "__main__":
|
|
132
|
+
"""example usage"""
|
|
133
|
+
import os
|
|
134
|
+
|
|
135
|
+
path = pathlib.Path(__file__).parents[1] / "cvat_labeling" / "example" / "coco.json"
|
|
136
|
+
|
|
137
|
+
coco_json_path = str(path)
|
|
138
|
+
coco_dir = os.path.dirname(coco_json_path)
|
|
139
|
+
coco_file_name = os.path.basename(coco_json_path)
|
|
140
|
+
coco_target_dir = os.path.join(os.path.dirname(coco_dir), "yolo_dataset")
|
|
141
|
+
os.makedirs(coco_target_dir, exist_ok=True)
|
|
142
|
+
|
|
143
|
+
create_yolo_dataset_from_coco_instances_dataset(coco_json_path, coco_target_dir, use_segmentation=True)
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from typing import List, Optional
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def view_coco_dataset(
|
|
6
|
+
labels_json_path: str, dataset_dir: Optional[str] = None, label_types: Optional[List[str]] = None
|
|
7
|
+
) -> None:
|
|
8
|
+
"""visualize a coco dataset in fiftyone"""
|
|
9
|
+
# lazy import because fiftyone is slow to import and makes the CLI slow.
|
|
10
|
+
import fiftyone as fo
|
|
11
|
+
|
|
12
|
+
if dataset_dir is None:
|
|
13
|
+
dataset_dir = os.path.dirname(labels_json_path)
|
|
14
|
+
if label_types is None or not label_types:
|
|
15
|
+
label_types = ["detections", "segmentations", "keypoints"]
|
|
16
|
+
else:
|
|
17
|
+
assert all([label_type in ["detections", "segmentations", "keypoints"] for label_type in label_types])
|
|
18
|
+
|
|
19
|
+
labels_json_path = os.path.realpath(labels_json_path)
|
|
20
|
+
|
|
21
|
+
dataset = fo.Dataset.from_dir(
|
|
22
|
+
dataset_type=fo.types.COCODetectionDataset,
|
|
23
|
+
label_types=label_types,
|
|
24
|
+
data_path=dataset_dir,
|
|
25
|
+
labels_path=labels_json_path,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
session = fo.launch_app(dataset)
|
|
29
|
+
session.wait()
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""merge 2 coco datasets into one"""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import pathlib
|
|
5
|
+
import shutil
|
|
6
|
+
|
|
7
|
+
import tqdm
|
|
8
|
+
from airo_dataset_tools.data_parsers.coco import CocoInstancesDataset
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def merge_coco_annotations(dataset1: CocoInstancesDataset, dataset2: CocoInstancesDataset) -> CocoInstancesDataset:
|
|
12
|
+
"""merge 2 coco annotations schemas. Categories and Annotations are assumed to be unique. Images can be present in both datasets.
|
|
13
|
+
|
|
14
|
+
Images will be checked for duplicates based on their file name.
|
|
15
|
+
|
|
16
|
+
Annotation IDs will be changed to avoid conflicts and their image IDs will be updated if needed."""
|
|
17
|
+
categories_1 = dataset1.categories
|
|
18
|
+
categories_2 = dataset2.categories
|
|
19
|
+
|
|
20
|
+
merged_categories = list(categories_1)
|
|
21
|
+
for category in categories_2:
|
|
22
|
+
if category.id not in [category.id for category in merged_categories]:
|
|
23
|
+
merged_categories.append(category)
|
|
24
|
+
else:
|
|
25
|
+
same_id_category = merged_categories[[category.id for category in merged_categories].index(category.id)]
|
|
26
|
+
if category != same_id_category:
|
|
27
|
+
raise ValueError(
|
|
28
|
+
"Categories with the same ID are not equal: " + str(category) + " " + str(same_id_category)
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
merged_images = dataset1.images
|
|
32
|
+
max_dataset_image_id = max([image.id for image in merged_images])
|
|
33
|
+
merged_image_paths = [image.file_name for image in merged_images]
|
|
34
|
+
|
|
35
|
+
dataset_2_image_id_mapping = {}
|
|
36
|
+
for image in dataset2.images:
|
|
37
|
+
if image.file_name not in merged_image_paths:
|
|
38
|
+
# create a new ID to avoid collisions
|
|
39
|
+
old_id = image.id
|
|
40
|
+
image.id = max_dataset_image_id + 1
|
|
41
|
+
max_dataset_image_id += 1
|
|
42
|
+
dataset_2_image_id_mapping[old_id] = image.id
|
|
43
|
+
merged_images.append(image)
|
|
44
|
+
merged_image_paths.append(image.file_name)
|
|
45
|
+
else:
|
|
46
|
+
# dataset already contained this image, so we need to remap the annotations
|
|
47
|
+
dataset_2_image_id_mapping[image.id] = merged_images[merged_image_paths.index(image.file_name)].id
|
|
48
|
+
|
|
49
|
+
merged_annotations = list(dataset1.annotations)
|
|
50
|
+
max_dataset_annotation_id = max([annotation.id for annotation in merged_annotations])
|
|
51
|
+
for annotation in dataset2.annotations:
|
|
52
|
+
if annotation.image_id in dataset_2_image_id_mapping.keys():
|
|
53
|
+
annotation.image_id = dataset_2_image_id_mapping[annotation.image_id]
|
|
54
|
+
annotation.id = max_dataset_annotation_id + 1
|
|
55
|
+
max_dataset_annotation_id += 1
|
|
56
|
+
merged_annotations.append(annotation)
|
|
57
|
+
|
|
58
|
+
merged_dataset = CocoInstancesDataset(
|
|
59
|
+
categories=merged_categories, images=merged_images, annotations=merged_annotations
|
|
60
|
+
)
|
|
61
|
+
return merged_dataset
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def merge_coco_image_folders(dataset1_base_dir: str, dataset2_base_dir: str, target_dir: str) -> None:
|
|
65
|
+
"""merge 2 image folders into one. Images will be copied to the target folder.
|
|
66
|
+
|
|
67
|
+
all base_dirs have following setup:
|
|
68
|
+
|
|
69
|
+
base_dir
|
|
70
|
+
--- images
|
|
71
|
+
------ image1.<>
|
|
72
|
+
------ image2.<>
|
|
73
|
+
<>.json // annotations with path relative to base_dir
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
dataset1_base_dir_path = pathlib.Path(dataset1_base_dir)
|
|
77
|
+
dataset2_base_dir_path = pathlib.Path(dataset2_base_dir)
|
|
78
|
+
target_dir_path = pathlib.Path(target_dir)
|
|
79
|
+
|
|
80
|
+
dataset1_image_paths = [image_path for image_path in dataset1_base_dir_path.iterdir()]
|
|
81
|
+
dataset2_image_paths = [image_path for image_path in dataset2_base_dir_path.iterdir()]
|
|
82
|
+
|
|
83
|
+
target_image_dir = target_dir_path / "images"
|
|
84
|
+
target_image_dir.mkdir(parents=True, exist_ok=True)
|
|
85
|
+
|
|
86
|
+
for image_path in tqdm.tqdm(
|
|
87
|
+
dataset1_image_paths, desc=f"copying images from {dataset1_base_dir_path.name} to {target_dir_path.name}"
|
|
88
|
+
):
|
|
89
|
+
shutil.copy(image_path, target_image_dir / image_path.name)
|
|
90
|
+
|
|
91
|
+
for image_path in tqdm.tqdm(
|
|
92
|
+
dataset2_image_paths, desc=f"copying images from {dataset2_base_dir_path.name} to {target_dir_path.name}"
|
|
93
|
+
):
|
|
94
|
+
if not (target_image_dir / image_path.name).exists():
|
|
95
|
+
shutil.copy(image_path, target_image_dir / image_path.name)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def merge_coco_datasets(json_path_1: str, json_path_2: str, target_json_path: str) -> None:
|
|
99
|
+
"""merge 2 coco datasets into one. Categories and Annotations are assumed to be unique. Images can be present in both datasets.
|
|
100
|
+
|
|
101
|
+
Images will be checked for duplicates based on their file name.
|
|
102
|
+
|
|
103
|
+
Annotation IDs will be changed to avoid conflicts and their image IDs will be updated if needed."""
|
|
104
|
+
|
|
105
|
+
image_path_1 = pathlib.Path(json_path_1).parent / "images"
|
|
106
|
+
image_path_2 = pathlib.Path(json_path_2).parent / "images"
|
|
107
|
+
|
|
108
|
+
merge_coco_image_folders(str(image_path_1), str(image_path_2), str(pathlib.Path(target_json_path).parent))
|
|
109
|
+
|
|
110
|
+
dataset1 = None
|
|
111
|
+
with open(json_path_1, "r") as f:
|
|
112
|
+
dataset1 = CocoInstancesDataset(**json.load(f))
|
|
113
|
+
|
|
114
|
+
dataset2 = None
|
|
115
|
+
with open(json_path_2, "r") as f:
|
|
116
|
+
dataset2 = CocoInstancesDataset(**json.load(f))
|
|
117
|
+
|
|
118
|
+
merged_dataset = merge_coco_annotations(dataset1, dataset2)
|
|
119
|
+
|
|
120
|
+
with open(target_json_path, "w") as f:
|
|
121
|
+
json.dump(merged_dataset.model_dump(exclude_none=True), f)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
if __name__ == "__main__":
|
|
125
|
+
json1 = (
|
|
126
|
+
"/home/tlips/Documents/airo-mono/airo-dataset-tools/test/test_data/lego-battery-resized/annotations_train.json"
|
|
127
|
+
)
|
|
128
|
+
json2 = (
|
|
129
|
+
"/home/tlips/Documents/airo-mono/airo-dataset-tools/test/test_data/lego-battery-resized/annotations_val.json"
|
|
130
|
+
)
|
|
131
|
+
target_json = "/home/tlips/Documents/airo-mono/airo-dataset-tools/test/test_data/lego-battery-resize-merged/annotations_merged.json"
|
|
132
|
+
merge_coco_datasets(json1, json2, target_json)
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
""" Split a COCO dataset into subsets"""
|
|
2
|
+
import json
|
|
3
|
+
import pathlib
|
|
4
|
+
import random
|
|
5
|
+
from typing import List, Optional
|
|
6
|
+
|
|
7
|
+
from airo_dataset_tools.data_parsers.coco import CocoImage, CocoInstanceAnnotation, CocoInstancesDataset
|
|
8
|
+
from pydantic import ValidationError
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def split_coco_dataset(
|
|
12
|
+
coco_dataset: CocoInstancesDataset,
|
|
13
|
+
split_ratios: List[float],
|
|
14
|
+
shuffle_before_splitting: bool = True,
|
|
15
|
+
shuffle_seed: int = 42,
|
|
16
|
+
) -> List[CocoInstancesDataset]:
|
|
17
|
+
"""Split a COCO dataset into subsets by splitting the images according to the specified relative ratios.
|
|
18
|
+
All annotations for an image will be placed in the same subset as the image.
|
|
19
|
+
|
|
20
|
+
Note that this does not guarantee the ratio of the annotations OR an equal class balance in each subset.
|
|
21
|
+
|
|
22
|
+
Ratios must sum to 1.0.
|
|
23
|
+
"""
|
|
24
|
+
ratio_sum = sum(split_ratios)
|
|
25
|
+
if abs(ratio_sum - 1.0) > 2e-2:
|
|
26
|
+
raise ValueError(f"Ratios must sum to 1.0. Ratios sum to {ratio_sum}.")
|
|
27
|
+
|
|
28
|
+
# split the images into 2 subsets (random or ordered)
|
|
29
|
+
images = coco_dataset.images
|
|
30
|
+
|
|
31
|
+
if shuffle_before_splitting:
|
|
32
|
+
# make shuffling reproducible
|
|
33
|
+
# use local random instance to not influence
|
|
34
|
+
# the global random state
|
|
35
|
+
rng = random.Random(shuffle_seed)
|
|
36
|
+
rng.shuffle(images)
|
|
37
|
+
|
|
38
|
+
image_splits: List[List[CocoImage]] = []
|
|
39
|
+
split_sizes = [round(ratio * len(images)) for ratio in split_ratios]
|
|
40
|
+
split_sizes[-1] = len(images) - sum(split_sizes[:-1]) # make sure the total number of images is correct
|
|
41
|
+
print(f"Split sizes: {split_sizes}")
|
|
42
|
+
|
|
43
|
+
for size in split_sizes:
|
|
44
|
+
image_splits.append(images[:size])
|
|
45
|
+
images = images[size:]
|
|
46
|
+
|
|
47
|
+
image_id_to_split_id = {}
|
|
48
|
+
for split_id, image_split in enumerate(image_splits):
|
|
49
|
+
for image in image_split:
|
|
50
|
+
image_id_to_split_id[image.id] = split_id
|
|
51
|
+
|
|
52
|
+
# gather the annotations for each subset
|
|
53
|
+
annotation_splits: List[List[CocoInstanceAnnotation]] = [[] for _ in range(len(split_ratios))]
|
|
54
|
+
for annotation in coco_dataset.annotations:
|
|
55
|
+
image_id = annotation.image_id
|
|
56
|
+
split_id = image_id_to_split_id[image_id]
|
|
57
|
+
# keep original image_ids and annotation_ids so that you could still reference the original dataset
|
|
58
|
+
annotation_splits[split_id].append(annotation)
|
|
59
|
+
|
|
60
|
+
# check if none of the annotation splits are empty:
|
|
61
|
+
for split_id, annotation_split in enumerate(annotation_splits):
|
|
62
|
+
if len(annotation_split) == 0:
|
|
63
|
+
raise ValueError(
|
|
64
|
+
f"Split {split_id} is empty, which is not allowed. Please use a larger dataset or a smaller split ratio."
|
|
65
|
+
)
|
|
66
|
+
# create a new COCO dataset for each subset
|
|
67
|
+
coco_dataset_splits: List[CocoInstancesDataset] = []
|
|
68
|
+
for annotation_split, image_split in zip(annotation_splits, image_splits):
|
|
69
|
+
coco_dataset_split = CocoInstancesDataset(
|
|
70
|
+
categories=coco_dataset.categories, images=image_split, annotations=annotation_split
|
|
71
|
+
)
|
|
72
|
+
coco_dataset_splits.append(coco_dataset_split)
|
|
73
|
+
|
|
74
|
+
return coco_dataset_splits
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def split_and_save_coco_dataset(
|
|
78
|
+
coco_json_path: str, split_ratios: List[float], shuffle_before_splitting: bool = True
|
|
79
|
+
) -> None:
|
|
80
|
+
"""Split a COCO dataset into subsets according to the specified relative ratios and save them to disk.
|
|
81
|
+
Images are split with their corresponding annotations. No guarantees on class balance or annotation ratios.
|
|
82
|
+
|
|
83
|
+
Ratios must sum to 1.0.
|
|
84
|
+
|
|
85
|
+
If two ratios are specified, the dataset will be split into two subsets. these will be called train/val by default.
|
|
86
|
+
If three ratios are specified, the dataset will be split into three subsets. these will be called train/val/test by default.
|
|
87
|
+
"""
|
|
88
|
+
split_names = ["train", "val", "test"]
|
|
89
|
+
if len(split_ratios) > len(split_names):
|
|
90
|
+
raise ValueError(f"Only {len(split_names)} splits are supported. {len(split_ratios)} splits were specified.")
|
|
91
|
+
|
|
92
|
+
coco_dataset: Optional[CocoInstancesDataset] = None
|
|
93
|
+
with open(coco_json_path, "r") as f:
|
|
94
|
+
try:
|
|
95
|
+
coco_dataset = CocoInstancesDataset(**json.load(f))
|
|
96
|
+
except ValidationError as e:
|
|
97
|
+
print(e)
|
|
98
|
+
raise ValueError("Could not load CocoInstancesDataset")
|
|
99
|
+
|
|
100
|
+
coco_dataset_splits = split_coco_dataset(coco_dataset, split_ratios, shuffle_before_splitting)
|
|
101
|
+
|
|
102
|
+
for split_id, coco_dataset_split in enumerate(coco_dataset_splits):
|
|
103
|
+
|
|
104
|
+
file_name = coco_json_path.replace(".json", f"_{split_names[split_id]}.json")
|
|
105
|
+
with open(file_name, "w") as f:
|
|
106
|
+
json.dump(coco_dataset_split.model_dump(exclude_none=True), f)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
if __name__ == "__main__":
|
|
110
|
+
json_path = pathlib.Path(__file__).parents[2] / "test" / "test_data" / "person_keypoints_val2017_small.json"
|
|
111
|
+
print(json_path)
|
|
112
|
+
split_and_save_coco_dataset(str(json_path), [0.5, 0.5])
|