airo-dataset-tools 2025.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. airo_dataset_tools-2025.4.0/PKG-INFO +24 -0
  2. airo_dataset_tools-2025.4.0/README.md +20 -0
  3. airo_dataset_tools-2025.4.0/airo_dataset_tools/__init__.py +5 -0
  4. airo_dataset_tools-2025.4.0/airo_dataset_tools/cli.py +136 -0
  5. airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/__init__.py +0 -0
  6. airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/change_coco_images_prefix.py +40 -0
  7. airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/coco_instances_to_yolo.py +143 -0
  8. airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/fiftyone_viewer.py +29 -0
  9. airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/merge_datasets.py +132 -0
  10. airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/split_dataset.py +112 -0
  11. airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/transform_dataset.py +218 -0
  12. airo_dataset_tools-2025.4.0/airo_dataset_tools/coco_tools/transforms.py +37 -0
  13. airo_dataset_tools-2025.4.0/airo_dataset_tools/cvat_labeling/__init__.py +0 -0
  14. airo_dataset_tools-2025.4.0/airo_dataset_tools/cvat_labeling/convert_cvat_to_coco.py +302 -0
  15. airo_dataset_tools-2025.4.0/airo_dataset_tools/cvat_labeling/load_xml_to_dict.py +24 -0
  16. airo_dataset_tools-2025.4.0/airo_dataset_tools/data_parsers/__init__.py +4 -0
  17. airo_dataset_tools-2025.4.0/airo_dataset_tools/data_parsers/camera_intrinsics.py +70 -0
  18. airo_dataset_tools-2025.4.0/airo_dataset_tools/data_parsers/coco.py +205 -0
  19. airo_dataset_tools-2025.4.0/airo_dataset_tools/data_parsers/cvat_images.py +142 -0
  20. airo_dataset_tools-2025.4.0/airo_dataset_tools/data_parsers/pose.py +58 -0
  21. airo_dataset_tools-2025.4.0/airo_dataset_tools/py.typed +0 -0
  22. airo_dataset_tools-2025.4.0/airo_dataset_tools/segmentation_mask_converter.py +186 -0
  23. airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/PKG-INFO +24 -0
  24. airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/SOURCES.txt +38 -0
  25. airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/dependency_links.txt +1 -0
  26. airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/entry_points.txt +2 -0
  27. airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/requires.txt +14 -0
  28. airo_dataset_tools-2025.4.0/airo_dataset_tools.egg-info/top_level.txt +1 -0
  29. airo_dataset_tools-2025.4.0/setup.cfg +4 -0
  30. airo_dataset_tools-2025.4.0/setup.py +35 -0
  31. airo_dataset_tools-2025.4.0/test/test_camera_intrinsics.py +54 -0
  32. airo_dataset_tools-2025.4.0/test/test_coco_create.py +181 -0
  33. airo_dataset_tools-2025.4.0/test/test_coco_load.py +101 -0
  34. airo_dataset_tools-2025.4.0/test/test_coco_merge.py +29 -0
  35. airo_dataset_tools-2025.4.0/test/test_coco_split.py +33 -0
  36. airo_dataset_tools-2025.4.0/test/test_cvat_images_load.py +18 -0
  37. airo_dataset_tools-2025.4.0/test/test_cvat_to_coco_conversion.py +8 -0
  38. airo_dataset_tools-2025.4.0/test/test_pillow_resize.py +9 -0
  39. airo_dataset_tools-2025.4.0/test/test_pose.py +48 -0
  40. airo_dataset_tools-2025.4.0/test/test_segmentation_mask.py +62 -0
@@ -0,0 +1,24 @@
1
+ Metadata-Version: 2.4
2
+ Name: airo-dataset-tools
3
+ Version: 2025.4.0
4
+ Summary: Scripts for loading and converting datasets for the Ghent University AI and Robotics Lab
5
+ Author: Victor-Louis De Gusseme
6
+ Author-email: victorlouisdg@gmail.com
7
+ Requires-Dist: numpy<2.0
8
+ Requires-Dist: pydantic>2.0.0
9
+ Requires-Dist: opencv-contrib-python==4.8.1.78
10
+ Requires-Dist: opencv-python-headless==4.8.1.78
11
+ Requires-Dist: pycocotools
12
+ Requires-Dist: xmltodict
13
+ Requires-Dist: tqdm
14
+ Requires-Dist: fiftyone
15
+ Requires-Dist: Pillow
16
+ Requires-Dist: types-Pillow
17
+ Requires-Dist: albumentations
18
+ Requires-Dist: click
19
+ Requires-Dist: airo-typing==2025.4.0
20
+ Requires-Dist: airo-spatial-algebra==2025.4.0
21
+ Dynamic: author
22
+ Dynamic: author-email
23
+ Dynamic: requires-dist
24
+ Dynamic: summary
@@ -0,0 +1,20 @@
1
+ # airo-dataset-tools
2
+ Tools for working with datasets.
3
+ They fall into two categories:
4
+
5
+ [**COCO related tools**](airo_dataset_tools/coco_tools/README.md):
6
+ * COCO dataset loading (and creation)
7
+ * FiftyOne visualisation
8
+ * Albumentation transforms
9
+ * COCO to YOLO conversion.
10
+ * CVAT labeling workflow
11
+
12
+ [**Data formats**](airo_dataset_tools/data_parsers/README.md):
13
+ * 3D poses
14
+ * Camera instrinsics
15
+
16
+
17
+ > [Pydantic](https://docs.pydantic.dev/latest/) is used heavily throughout this package.
18
+ It allows you to easily create Python objects that can be saved and loaded to and from JSON files.
19
+
20
+ [**CVAT labeling workflow & tools**](airo_dataset_tools/cvat_labeling/readme.md)
@@ -0,0 +1,5 @@
1
+ # init files are still required in modern python!
2
+ # https://peps.python.org/pep-0420/ introduced implicit namespace packages
3
+ # but for building and many toolings, you still need to have __init__ files (at least in the root of the package).
4
+ # e.g. if you remove this init file and try to build with pip install .
5
+ # you won't be able to import the dummy module.
@@ -0,0 +1,136 @@
1
+ """CLI interface for this package"""
2
+
3
+ import json
4
+ import os
5
+ from typing import List, Optional
6
+
7
+ import click
8
+ from airo_dataset_tools.coco_tools.change_coco_images_prefix import change_coco_json_image_prefix
9
+ from airo_dataset_tools.coco_tools.coco_instances_to_yolo import create_yolo_dataset_from_coco_instances_dataset
10
+ from airo_dataset_tools.coco_tools.fiftyone_viewer import view_coco_dataset
11
+ from airo_dataset_tools.coco_tools.merge_datasets import merge_coco_datasets
12
+ from airo_dataset_tools.coco_tools.split_dataset import split_and_save_coco_dataset
13
+ from airo_dataset_tools.coco_tools.transform_dataset import resize_coco_dataset
14
+ from airo_dataset_tools.cvat_labeling.convert_cvat_to_coco import cvat_image_to_coco
15
+
16
+
17
+ @click.group()
18
+ def cli() -> None:
19
+ """CLI entrypoint for airo-dataset-tools"""
20
+
21
+
22
+ @cli.command(name="fiftyone-coco-viewer") # no help, takes the docstring of the function.
23
+ @click.argument("annotations-json-path", type=click.Path(exists=True))
24
+ @click.option(
25
+ "--dataset-dir",
26
+ required=False,
27
+ type=click.Path(exists=True),
28
+ help="optional directory relative to which the image paths in the coco dataset are specified",
29
+ )
30
+ @click.option(
31
+ "--label-types",
32
+ "-l",
33
+ multiple=True,
34
+ type=click.Choice(["detections", "segmentations", "keypoints"]),
35
+ help="add an argument for each label type you want to load (default: all)",
36
+ )
37
+ def view_coco_dataset_cli(
38
+ annotations_json_path: str, dataset_dir: str, label_types: Optional[List[str]] = None
39
+ ) -> None:
40
+ """Explore COCO dataset with FiftyOne"""
41
+ view_coco_dataset(annotations_json_path, dataset_dir, label_types)
42
+
43
+
44
+ @cli.command(name="convert-cvat-to-coco-keypoints")
45
+ @click.argument("cvat_xml_file", type=str, required=True)
46
+ @click.argument("coco-categories-json-file", type=str, required=True)
47
+ @click.option("--add-bbox", is_flag=True, default=False, help="include bounding box in coco annotations")
48
+ @click.option("--add-segmentation", is_flag=True, default=False, help="include segmentation in coco annotations")
49
+ def convert_cvat_to_coco_cli(
50
+ cvat_xml_file: str, coco_categories_json_file: str, add_bbox: bool, add_segmentation: bool
51
+ ) -> None:
52
+ """Convert CVAT XML to COCO keypoints json according to specified coco categories"""
53
+ coco = cvat_image_to_coco(
54
+ cvat_xml_file, coco_categories_json_file, add_bbox=add_bbox, add_segmentation=add_segmentation
55
+ )
56
+ path = os.path.dirname(cvat_xml_file)
57
+ filename = os.path.basename(cvat_xml_file)
58
+ path = os.path.join(path, filename.split(".")[0] + ".json")
59
+ with open(path, "w") as file:
60
+ json.dump(coco, file)
61
+
62
+
63
+ @cli.command(name="resize-coco-dataset")
64
+ @click.argument("annotations-json-path", type=click.Path(exists=True))
65
+ @click.option("--width", type=int, required=True)
66
+ @click.option("--height", type=int, required=True)
67
+ @click.option("--target-dataset-dir", type=str, required=False)
68
+ def resize_coco_dataset_cli(
69
+ annotations_json_path: str, width: int, height: int, target_dataset_dir: Optional[str]
70
+ ) -> None:
71
+ """Resize a COCO dataset. Will create a new directory with the resized dataset at the specified target_dataset_dir.
72
+ Dataset is assumed to be
73
+ /dir
74
+ annotations.json # contains relative paths w.r.t. /dir
75
+ ...
76
+ """
77
+ resize_coco_dataset(annotations_json_path, width, height, target_dataset_dir=target_dataset_dir)
78
+
79
+
80
+ @cli.command(name="coco-instances-to-yolo")
81
+ @click.option("--coco-json", type=str)
82
+ @click.option("--target-dir", type=str)
83
+ @click.option("--use-segmentation", is_flag=True)
84
+ def coco_intances_to_yolo(coco_json: str, target_dir: str, use_segmentation: bool) -> None:
85
+ """Create a YOLO detections/segmentations dataset from a coco instances dataset"""
86
+ print(f"converting coco instances dataset {coco_json} to yolo dataset {target_dir}")
87
+ create_yolo_dataset_from_coco_instances_dataset(coco_json, target_dir, use_segmentation=use_segmentation)
88
+
89
+
90
+ @cli.command(name="split-coco-dataset")
91
+ @click.argument("json-path", type=click.Path(exists=True))
92
+ @click.option("--split-ratio", type=float, multiple=True, required=True)
93
+ @click.option("--shuffle-before-splitting", is_flag=True, default=True)
94
+ def split_coco_dataset_cli(json_path: str, split_ratio: List[float], shuffle_before_splitting: bool) -> None:
95
+ """Split a COCO dataset into subsets according to the specified relative ratios and save them to disk.
96
+ Images are split with their corresponding annotations. No guarantees on class balance or annotation ratios.
97
+
98
+ If two ratios are specified, the dataset will be split into two subsets. these will be called train/val by default.
99
+ If three ratios are specified, the dataset will be split into three subsets. these will be called train/val/test by default.
100
+
101
+ e.g. split-coco-dataset <path> --split-ratio 0.8 --split-ratio 0.2
102
+ """
103
+ split_and_save_coco_dataset(json_path, split_ratio, shuffle_before_splitting=shuffle_before_splitting)
104
+
105
+
106
+ @cli.command(name="change-coco-images-prefix")
107
+ @click.argument("coco-json", type=click.Path(exists=True))
108
+ @click.option("--current-prefix", type=str, required=True)
109
+ @click.option("--new-prefix", type=str, required=True)
110
+ @click.option("--target-json-path", type=str)
111
+ def change_coco_images_prefix_cli(
112
+ coco_json: str, current_prefix: str, new_prefix: str, target_json_path: Optional[str] = None
113
+ ) -> None:
114
+ """change the prefix of images in a coco dataset."""
115
+ return change_coco_json_image_prefix(coco_json, current_prefix, new_prefix, target_json_path)
116
+
117
+
118
+ @cli.command(name="merge-coco-datasets")
119
+ @click.argument("coco-json-1", type=click.Path(exists=True))
120
+ @click.argument("coco-json-2", type=click.Path(exists=True))
121
+ @click.option(
122
+ "--target-json-path",
123
+ type=str,
124
+ help="optional path to save the merged dataset to. If none is provided, a new directory will be created in the parent directory of coco-json-1 ",
125
+ )
126
+ def merge_coco_datasets_cli(coco_json_1: str, coco_json_2: str, target_json_path: Optional[str] = None) -> None:
127
+ """merge two coco datasets into a single dataset."""
128
+ if not target_json_path:
129
+ target_json_path = os.path.join(os.path.dirname(coco_json_1), "merged")
130
+ os.makedirs(target_json_path, exist_ok=True)
131
+ target_json_path = os.path.join(target_json_path, "annotations.json")
132
+ return merge_coco_datasets(coco_json_1, coco_json_2, target_json_path)
133
+
134
+
135
+ if __name__ == "__main__":
136
+ cli()
@@ -0,0 +1,40 @@
1
+ import json
2
+ from typing import Optional
3
+
4
+ from airo_dataset_tools.data_parsers.coco import CocoInstancesDataset
5
+
6
+
7
+ def change_coco_dataset_image_prefix(
8
+ coco_dataset: CocoInstancesDataset, current_prefix: str, target_prefix: str
9
+ ) -> CocoInstancesDataset:
10
+ """Change the prefix of the image paths in a COCO dataset. Can be used to change the base directory of the images in a dataset.
11
+
12
+ Args:
13
+ coco_dataset: the dataset to modify
14
+ current_prefix: the current prefix of the image paths in the dataset
15
+ target_prefix: the target prefix of the image paths in the dataset
16
+
17
+ e.g.
18
+
19
+ if the images are currently relative to the image folder inside the dataset and you want to make them relative to the dataset folder:
20
+ change_coco_image_base_dir(coco_dataset, "", "images/")
21
+ """
22
+
23
+ for image in coco_dataset.images:
24
+ image.file_name = image.file_name.removeprefix(current_prefix)
25
+ image.file_name = target_prefix + image.file_name
26
+
27
+ return coco_dataset
28
+
29
+
30
+ def change_coco_json_image_prefix(
31
+ coco_json_file: str, current_prefix: str, target_prefix: str, target_json_file: Optional[str]
32
+ ) -> None:
33
+ with open(coco_json_file, "r") as f:
34
+ coco_dataset = CocoInstancesDataset(**json.load(f))
35
+
36
+ coco_dataset = change_coco_dataset_image_prefix(coco_dataset, current_prefix, target_prefix)
37
+ if target_json_file is None:
38
+ target_json_file = coco_json_file
39
+ with open(target_json_file, "w") as f:
40
+ json.dump(coco_dataset.model_dump(), f)
@@ -0,0 +1,143 @@
1
+ import json
2
+ import pathlib
3
+ from collections import defaultdict
4
+
5
+ import cv2
6
+ import tqdm
7
+ from airo_dataset_tools.data_parsers.coco import CocoInstancesDataset
8
+ from airo_dataset_tools.segmentation_mask_converter import BinarySegmentationMask
9
+
10
+
11
+ def create_yolo_dataset_from_coco_instances_dataset(
12
+ coco_dataset_json_path: str, target_directory: str, use_segmentation: bool = False
13
+ ) -> None:
14
+ """Converts a coco dataset to a yolo dataset, either segmentations or detections
15
+
16
+
17
+ coco dataset is expected:
18
+ dataset/
19
+ images/
20
+ <>.json # containing paths relative to /dataset
21
+
22
+ yolo dataset format will be
23
+
24
+ dataset/
25
+ images/
26
+ labels/
27
+ <>.txt # paths as in coco dataset /images subdir.
28
+
29
+ # ultralytics uses an additional <>.json, this has to be created manually..
30
+ # we do export a .names file, as in the darknet yolo format.
31
+ # each line contains a class name, the line number is the class id
32
+ obj.names
33
+ Args:
34
+ coco_dataset_json_path: _description_
35
+ target_directory: _description_
36
+ use_segmentation: create a segmentation dataset instead of a detection dataset (requires segmentation annotations)
37
+
38
+
39
+ quiz question: how many people would have written a similar converter in the past?...
40
+
41
+ """
42
+ _coco_dataset_json_path = pathlib.Path(coco_dataset_json_path)
43
+ _target_directory = pathlib.Path(target_directory)
44
+
45
+ # load the json file into the coco dataset object
46
+ annotations = json.load(open(_coco_dataset_json_path, "r"))
47
+ coco_dataset = CocoInstancesDataset(**annotations)
48
+
49
+ target_dir = pathlib.Path(_target_directory)
50
+ target_dir.mkdir(parents=True, exist_ok=True)
51
+ image_dir = target_dir / "images"
52
+ label_dir = target_dir / "labels"
53
+ image_dir.mkdir(parents=True, exist_ok=True)
54
+ label_dir.mkdir(parents=True, exist_ok=True)
55
+
56
+ # create the yolo dataset
57
+ annotations = coco_dataset.annotations
58
+ image_id_to_image = {image.id: image for image in coco_dataset.images}
59
+ image_id_to_annotations = defaultdict(list)
60
+ for annotation in annotations:
61
+ image_id_to_annotations[annotation.image_id].append(annotation)
62
+
63
+ category_id_to_category = {category.id: category for category in coco_dataset.categories}
64
+ # sort by original COCO ID, but whereas COCO does not care about ID ranges, YOLO expects them to be in the range 0..N-1
65
+ # so we sort them by ID and then use the index as the new ID
66
+ yolo_category_index = list(sorted(category_id_to_category.values(), key=lambda category: category.id))
67
+
68
+ for image_id, annotations in tqdm.tqdm(image_id_to_annotations.items()):
69
+ coco_image = image_id_to_image[image_id]
70
+ image_path = pathlib.Path(coco_image.file_name)
71
+ if not image_path.is_absolute():
72
+ image_path = _coco_dataset_json_path.parent / image_path
73
+
74
+ relative_image_path = image_path.relative_to(_coco_dataset_json_path.parent)
75
+
76
+ # ultralytics parser finds the 'latest' occurance of 'images' in the dataset path,
77
+ # so we need to replace any occurance of 'image' with 'img'
78
+ # https://github.com/ultralytics/ultralytics/issues/3581
79
+
80
+ relative_image_path_str = str(relative_image_path).replace("image", "img")
81
+ relative_image_path = pathlib.Path(relative_image_path_str)
82
+
83
+ image = cv2.imread(str(image_path))
84
+ height, width, _ = image.shape
85
+
86
+ label_path = label_dir / f"{relative_image_path.with_suffix('')}.txt"
87
+ label_path.parent.mkdir(parents=True, exist_ok=True)
88
+
89
+ with open(label_path, "w") as file:
90
+ for annotation in annotations:
91
+ category = category_id_to_category[annotation.category_id]
92
+ yolo_id = yolo_category_index.index(category)
93
+ if use_segmentation:
94
+ segmentation = annotation.segmentation
95
+ # convert to **single** polygon
96
+ segmentation = BinarySegmentationMask.from_coco_segmentation_mask(segmentation, width, height)
97
+ segmentation = segmentation.as_single_polygon
98
+
99
+ if segmentation is None:
100
+ # should actually never happen as each annotation is assumed to have a segmentation if you pass use_segmentation=True
101
+ # but we filter it for convenience to deal with edge cases
102
+ print(f"skipping annotation for image {image_path}, as it has no segmentation")
103
+ continue
104
+
105
+ file.write(f"{yolo_id}")
106
+ for (x, y) in zip(segmentation[0::2], segmentation[1::2]):
107
+ file.write(f" {x/width} {y/height}")
108
+ file.write("\n")
109
+
110
+ else:
111
+ x, y, w, h = annotation.bbox
112
+ x_center = x + w / 2
113
+ y_center = y + h / 2
114
+ x_center /= width
115
+ y_center /= height
116
+ w /= width
117
+ h /= height
118
+ file.write(f"{yolo_id} {x_center} {y_center} {w} {h}\n")
119
+
120
+ image_target_path = image_dir / relative_image_path
121
+ image_target_path.parent.mkdir(parents=True, exist_ok=True)
122
+ cv2.imwrite(str(image_target_path), image)
123
+
124
+ # create the obj.names file
125
+ with open(target_dir / "obj.names", "w") as file:
126
+ # sort categories by id
127
+ for category in yolo_category_index:
128
+ file.write(f"{category.name}\n")
129
+
130
+
131
+ if __name__ == "__main__":
132
+ """example usage"""
133
+ import os
134
+
135
+ path = pathlib.Path(__file__).parents[1] / "cvat_labeling" / "example" / "coco.json"
136
+
137
+ coco_json_path = str(path)
138
+ coco_dir = os.path.dirname(coco_json_path)
139
+ coco_file_name = os.path.basename(coco_json_path)
140
+ coco_target_dir = os.path.join(os.path.dirname(coco_dir), "yolo_dataset")
141
+ os.makedirs(coco_target_dir, exist_ok=True)
142
+
143
+ create_yolo_dataset_from_coco_instances_dataset(coco_json_path, coco_target_dir, use_segmentation=True)
@@ -0,0 +1,29 @@
1
+ import os
2
+ from typing import List, Optional
3
+
4
+
5
+ def view_coco_dataset(
6
+ labels_json_path: str, dataset_dir: Optional[str] = None, label_types: Optional[List[str]] = None
7
+ ) -> None:
8
+ """visualize a coco dataset in fiftyone"""
9
+ # lazy import because fiftyone is slow to import and makes the CLI slow.
10
+ import fiftyone as fo
11
+
12
+ if dataset_dir is None:
13
+ dataset_dir = os.path.dirname(labels_json_path)
14
+ if label_types is None or not label_types:
15
+ label_types = ["detections", "segmentations", "keypoints"]
16
+ else:
17
+ assert all([label_type in ["detections", "segmentations", "keypoints"] for label_type in label_types])
18
+
19
+ labels_json_path = os.path.realpath(labels_json_path)
20
+
21
+ dataset = fo.Dataset.from_dir(
22
+ dataset_type=fo.types.COCODetectionDataset,
23
+ label_types=label_types,
24
+ data_path=dataset_dir,
25
+ labels_path=labels_json_path,
26
+ )
27
+
28
+ session = fo.launch_app(dataset)
29
+ session.wait()
@@ -0,0 +1,132 @@
1
+ """merge 2 coco datasets into one"""
2
+
3
+ import json
4
+ import pathlib
5
+ import shutil
6
+
7
+ import tqdm
8
+ from airo_dataset_tools.data_parsers.coco import CocoInstancesDataset
9
+
10
+
11
+ def merge_coco_annotations(dataset1: CocoInstancesDataset, dataset2: CocoInstancesDataset) -> CocoInstancesDataset:
12
+ """merge 2 coco annotations schemas. Categories and Annotations are assumed to be unique. Images can be present in both datasets.
13
+
14
+ Images will be checked for duplicates based on their file name.
15
+
16
+ Annotation IDs will be changed to avoid conflicts and their image IDs will be updated if needed."""
17
+ categories_1 = dataset1.categories
18
+ categories_2 = dataset2.categories
19
+
20
+ merged_categories = list(categories_1)
21
+ for category in categories_2:
22
+ if category.id not in [category.id for category in merged_categories]:
23
+ merged_categories.append(category)
24
+ else:
25
+ same_id_category = merged_categories[[category.id for category in merged_categories].index(category.id)]
26
+ if category != same_id_category:
27
+ raise ValueError(
28
+ "Categories with the same ID are not equal: " + str(category) + " " + str(same_id_category)
29
+ )
30
+
31
+ merged_images = dataset1.images
32
+ max_dataset_image_id = max([image.id for image in merged_images])
33
+ merged_image_paths = [image.file_name for image in merged_images]
34
+
35
+ dataset_2_image_id_mapping = {}
36
+ for image in dataset2.images:
37
+ if image.file_name not in merged_image_paths:
38
+ # create a new ID to avoid collisions
39
+ old_id = image.id
40
+ image.id = max_dataset_image_id + 1
41
+ max_dataset_image_id += 1
42
+ dataset_2_image_id_mapping[old_id] = image.id
43
+ merged_images.append(image)
44
+ merged_image_paths.append(image.file_name)
45
+ else:
46
+ # dataset already contained this image, so we need to remap the annotations
47
+ dataset_2_image_id_mapping[image.id] = merged_images[merged_image_paths.index(image.file_name)].id
48
+
49
+ merged_annotations = list(dataset1.annotations)
50
+ max_dataset_annotation_id = max([annotation.id for annotation in merged_annotations])
51
+ for annotation in dataset2.annotations:
52
+ if annotation.image_id in dataset_2_image_id_mapping.keys():
53
+ annotation.image_id = dataset_2_image_id_mapping[annotation.image_id]
54
+ annotation.id = max_dataset_annotation_id + 1
55
+ max_dataset_annotation_id += 1
56
+ merged_annotations.append(annotation)
57
+
58
+ merged_dataset = CocoInstancesDataset(
59
+ categories=merged_categories, images=merged_images, annotations=merged_annotations
60
+ )
61
+ return merged_dataset
62
+
63
+
64
+ def merge_coco_image_folders(dataset1_base_dir: str, dataset2_base_dir: str, target_dir: str) -> None:
65
+ """merge 2 image folders into one. Images will be copied to the target folder.
66
+
67
+ all base_dirs have following setup:
68
+
69
+ base_dir
70
+ --- images
71
+ ------ image1.<>
72
+ ------ image2.<>
73
+ <>.json // annotations with path relative to base_dir
74
+ """
75
+
76
+ dataset1_base_dir_path = pathlib.Path(dataset1_base_dir)
77
+ dataset2_base_dir_path = pathlib.Path(dataset2_base_dir)
78
+ target_dir_path = pathlib.Path(target_dir)
79
+
80
+ dataset1_image_paths = [image_path for image_path in dataset1_base_dir_path.iterdir()]
81
+ dataset2_image_paths = [image_path for image_path in dataset2_base_dir_path.iterdir()]
82
+
83
+ target_image_dir = target_dir_path / "images"
84
+ target_image_dir.mkdir(parents=True, exist_ok=True)
85
+
86
+ for image_path in tqdm.tqdm(
87
+ dataset1_image_paths, desc=f"copying images from {dataset1_base_dir_path.name} to {target_dir_path.name}"
88
+ ):
89
+ shutil.copy(image_path, target_image_dir / image_path.name)
90
+
91
+ for image_path in tqdm.tqdm(
92
+ dataset2_image_paths, desc=f"copying images from {dataset2_base_dir_path.name} to {target_dir_path.name}"
93
+ ):
94
+ if not (target_image_dir / image_path.name).exists():
95
+ shutil.copy(image_path, target_image_dir / image_path.name)
96
+
97
+
98
+ def merge_coco_datasets(json_path_1: str, json_path_2: str, target_json_path: str) -> None:
99
+ """merge 2 coco datasets into one. Categories and Annotations are assumed to be unique. Images can be present in both datasets.
100
+
101
+ Images will be checked for duplicates based on their file name.
102
+
103
+ Annotation IDs will be changed to avoid conflicts and their image IDs will be updated if needed."""
104
+
105
+ image_path_1 = pathlib.Path(json_path_1).parent / "images"
106
+ image_path_2 = pathlib.Path(json_path_2).parent / "images"
107
+
108
+ merge_coco_image_folders(str(image_path_1), str(image_path_2), str(pathlib.Path(target_json_path).parent))
109
+
110
+ dataset1 = None
111
+ with open(json_path_1, "r") as f:
112
+ dataset1 = CocoInstancesDataset(**json.load(f))
113
+
114
+ dataset2 = None
115
+ with open(json_path_2, "r") as f:
116
+ dataset2 = CocoInstancesDataset(**json.load(f))
117
+
118
+ merged_dataset = merge_coco_annotations(dataset1, dataset2)
119
+
120
+ with open(target_json_path, "w") as f:
121
+ json.dump(merged_dataset.model_dump(exclude_none=True), f)
122
+
123
+
124
+ if __name__ == "__main__":
125
+ json1 = (
126
+ "/home/tlips/Documents/airo-mono/airo-dataset-tools/test/test_data/lego-battery-resized/annotations_train.json"
127
+ )
128
+ json2 = (
129
+ "/home/tlips/Documents/airo-mono/airo-dataset-tools/test/test_data/lego-battery-resized/annotations_val.json"
130
+ )
131
+ target_json = "/home/tlips/Documents/airo-mono/airo-dataset-tools/test/test_data/lego-battery-resize-merged/annotations_merged.json"
132
+ merge_coco_datasets(json1, json2, target_json)
@@ -0,0 +1,112 @@
1
+ """ Split a COCO dataset into subsets"""
2
+ import json
3
+ import pathlib
4
+ import random
5
+ from typing import List, Optional
6
+
7
+ from airo_dataset_tools.data_parsers.coco import CocoImage, CocoInstanceAnnotation, CocoInstancesDataset
8
+ from pydantic import ValidationError
9
+
10
+
11
+ def split_coco_dataset(
12
+ coco_dataset: CocoInstancesDataset,
13
+ split_ratios: List[float],
14
+ shuffle_before_splitting: bool = True,
15
+ shuffle_seed: int = 42,
16
+ ) -> List[CocoInstancesDataset]:
17
+ """Split a COCO dataset into subsets by splitting the images according to the specified relative ratios.
18
+ All annotations for an image will be placed in the same subset as the image.
19
+
20
+ Note that this does not guarantee the ratio of the annotations OR an equal class balance in each subset.
21
+
22
+ Ratios must sum to 1.0.
23
+ """
24
+ ratio_sum = sum(split_ratios)
25
+ if abs(ratio_sum - 1.0) > 2e-2:
26
+ raise ValueError(f"Ratios must sum to 1.0. Ratios sum to {ratio_sum}.")
27
+
28
+ # split the images into 2 subsets (random or ordered)
29
+ images = coco_dataset.images
30
+
31
+ if shuffle_before_splitting:
32
+ # make shuffling reproducible
33
+ # use local random instance to not influence
34
+ # the global random state
35
+ rng = random.Random(shuffle_seed)
36
+ rng.shuffle(images)
37
+
38
+ image_splits: List[List[CocoImage]] = []
39
+ split_sizes = [round(ratio * len(images)) for ratio in split_ratios]
40
+ split_sizes[-1] = len(images) - sum(split_sizes[:-1]) # make sure the total number of images is correct
41
+ print(f"Split sizes: {split_sizes}")
42
+
43
+ for size in split_sizes:
44
+ image_splits.append(images[:size])
45
+ images = images[size:]
46
+
47
+ image_id_to_split_id = {}
48
+ for split_id, image_split in enumerate(image_splits):
49
+ for image in image_split:
50
+ image_id_to_split_id[image.id] = split_id
51
+
52
+ # gather the annotations for each subset
53
+ annotation_splits: List[List[CocoInstanceAnnotation]] = [[] for _ in range(len(split_ratios))]
54
+ for annotation in coco_dataset.annotations:
55
+ image_id = annotation.image_id
56
+ split_id = image_id_to_split_id[image_id]
57
+ # keep original image_ids and annotation_ids so that you could still reference the original dataset
58
+ annotation_splits[split_id].append(annotation)
59
+
60
+ # check if none of the annotation splits are empty:
61
+ for split_id, annotation_split in enumerate(annotation_splits):
62
+ if len(annotation_split) == 0:
63
+ raise ValueError(
64
+ f"Split {split_id} is empty, which is not allowed. Please use a larger dataset or a smaller split ratio."
65
+ )
66
+ # create a new COCO dataset for each subset
67
+ coco_dataset_splits: List[CocoInstancesDataset] = []
68
+ for annotation_split, image_split in zip(annotation_splits, image_splits):
69
+ coco_dataset_split = CocoInstancesDataset(
70
+ categories=coco_dataset.categories, images=image_split, annotations=annotation_split
71
+ )
72
+ coco_dataset_splits.append(coco_dataset_split)
73
+
74
+ return coco_dataset_splits
75
+
76
+
77
+ def split_and_save_coco_dataset(
78
+ coco_json_path: str, split_ratios: List[float], shuffle_before_splitting: bool = True
79
+ ) -> None:
80
+ """Split a COCO dataset into subsets according to the specified relative ratios and save them to disk.
81
+ Images are split with their corresponding annotations. No guarantees on class balance or annotation ratios.
82
+
83
+ Ratios must sum to 1.0.
84
+
85
+ If two ratios are specified, the dataset will be split into two subsets. these will be called train/val by default.
86
+ If three ratios are specified, the dataset will be split into three subsets. these will be called train/val/test by default.
87
+ """
88
+ split_names = ["train", "val", "test"]
89
+ if len(split_ratios) > len(split_names):
90
+ raise ValueError(f"Only {len(split_names)} splits are supported. {len(split_ratios)} splits were specified.")
91
+
92
+ coco_dataset: Optional[CocoInstancesDataset] = None
93
+ with open(coco_json_path, "r") as f:
94
+ try:
95
+ coco_dataset = CocoInstancesDataset(**json.load(f))
96
+ except ValidationError as e:
97
+ print(e)
98
+ raise ValueError("Could not load CocoInstancesDataset")
99
+
100
+ coco_dataset_splits = split_coco_dataset(coco_dataset, split_ratios, shuffle_before_splitting)
101
+
102
+ for split_id, coco_dataset_split in enumerate(coco_dataset_splits):
103
+
104
+ file_name = coco_json_path.replace(".json", f"_{split_names[split_id]}.json")
105
+ with open(file_name, "w") as f:
106
+ json.dump(coco_dataset_split.model_dump(exclude_none=True), f)
107
+
108
+
109
+ if __name__ == "__main__":
110
+ json_path = pathlib.Path(__file__).parents[2] / "test" / "test_data" / "person_keypoints_val2017_small.json"
111
+ print(json_path)
112
+ split_and_save_coco_dataset(str(json_path), [0.5, 0.5])