nvs2colmap 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/PKG-INFO +34 -10
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/README.md +33 -9
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/colmap.py +0 -1
- nvs2colmap-0.3.0/nvs2colmap/dynamic3dgs/__main__.py +130 -0
- nvs2colmap-0.3.0/nvs2colmap/dynamic3dgs/camera_meta.py +98 -0
- nvs2colmap-0.3.0/nvs2colmap/dynamic3dgs/link_frames.py +52 -0
- nvs2colmap-0.3.0/nvs2colmap/extract_videos.py +117 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap.egg-info/PKG-INFO +34 -10
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap.egg-info/SOURCES.txt +8 -2
- nvs2colmap-0.3.0/nvs2colmap.egg-info/entry_points.txt +4 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/pyproject.toml +3 -1
- nvs2colmap-0.2.0/nvs2colmap/stnerf/__init__.py +0 -0
- nvs2colmap-0.2.0/nvs2colmap.egg-info/entry_points.txt +0 -2
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/LICENSE +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/__init__.py +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/n3dv/__main__.py +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/n3dv/extract_videos.py +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/n3dv/poses_bounds.py +0 -0
- {nvs2colmap-0.2.0/nvs2colmap/dynamic3dgs → nvs2colmap-0.3.0/nvs2colmap/stnerf}/__init__.py +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/utils/__init__.py +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/utils/colmap.py +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/utils/ffmpeg.py +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/utils/rotation.py +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/write_model.py +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap.egg-info/dependency_links.txt +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap.egg-info/requires.txt +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap.egg-info/top_level.txt +0 -0
- {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: nvs2colmap
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Utilities for converting novel view synthesis datasets to COLMAP format.
|
|
5
5
|
Author-email: Howard Yin <yindaheng98@gmail.com>
|
|
6
6
|
Maintainer-email: Howard Yin <yindaheng98@gmail.com>
|
|
@@ -34,6 +34,8 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
|
|
|
34
34
|
|
|
35
35
|
- **Neural 3D Video Dataset**: scenes with `poses_bounds.npy` and one `mp4`
|
|
36
36
|
file per camera. See `nvs2colmap/n3dv/README.md`.
|
|
37
|
+
- **Dynamic 3D Gaussians**: scenes with `train_meta.json`, `test_meta.json`,
|
|
38
|
+
images in `ims/`, and masks in `seg/`. See `nvs2colmap/dynamic3dgs/README.md`.
|
|
37
39
|
|
|
38
40
|
## Supported Datasets
|
|
39
41
|
|
|
@@ -48,6 +50,10 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
|
|
|
48
50
|
[liuyubian/Robo360](https://huggingface.co/datasets/liuyubian/Robo360),
|
|
49
51
|
paper
|
|
50
52
|
[Robo360: A 3D Omnispective Multi-Material Robotic Manipulation Dataset](https://arxiv.org/abs/2312.06686).
|
|
53
|
+
- **Dynamic 3D Gaussians**: dataset
|
|
54
|
+
[JonathonLuiten/Dynamic3DGaussians](https://github.com/JonathonLuiten/Dynamic3DGaussians),
|
|
55
|
+
paper
|
|
56
|
+
[Dynamic 3D Gaussians: Tracking by Persistent Dynamic View Synthesis](https://arxiv.org/abs/2308.09713).
|
|
51
57
|
|
|
52
58
|
## Quick Start
|
|
53
59
|
|
|
@@ -88,18 +94,36 @@ Run the full COLMAP pipeline for each frame:
|
|
|
88
94
|
```bash
|
|
89
95
|
python -m nvs2colmap.n3dv \
|
|
90
96
|
--path data/Robo360/xarm6_gold_rope_in_basket_2 \
|
|
91
|
-
--ffmpeg
|
|
92
|
-
--ffprobe
|
|
97
|
+
--ffmpeg ffmpeg \
|
|
98
|
+
--ffprobe ffprobe \
|
|
93
99
|
--video-extension MP4 \
|
|
94
100
|
--n-frames 1 \
|
|
95
101
|
--use-colmap \
|
|
96
|
-
--colmap-executable
|
|
102
|
+
--colmap-executable colmap \
|
|
97
103
|
--colmap-use-gpu 1
|
|
98
104
|
```
|
|
99
105
|
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
+
Link a Dynamic 3D Gaussians scene, including its masks, and write per-frame
|
|
107
|
+
COLMAP text models:
|
|
108
|
+
|
|
109
|
+
```bash
|
|
110
|
+
python -m nvs2colmap.dynamic3dgs \
|
|
111
|
+
--path data/basketball \
|
|
112
|
+
--n-frames 150
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
For Neural 3D Video scenes, decoded frames are written to `frame*/images` by
|
|
116
|
+
default, and the command also writes `frame*/sparse/0` text models. With
|
|
117
|
+
`--use-colmap`, decoded frames are written to `frame*/input`, and each frame
|
|
118
|
+
additionally gets the standard COLMAP outputs such as `distorted/`, `images/`,
|
|
119
|
+
`sparse/`, and `stereo/`. When `--start-number N` is provided, decoding starts
|
|
120
|
+
from source video frame `N`, and the generated folders/images are also numbered
|
|
121
|
+
from `N`.
|
|
122
|
+
|
|
123
|
+
For Dynamic 3D Gaussians scenes, images from both `train_meta.json` and
|
|
124
|
+
`test_meta.json` are hardlinked into `frame*/images`, and included masks are
|
|
125
|
+
hardlinked into `frame*/image_masks` (`cam01.jpg` pairs with `cam01.jpg.png`).
|
|
126
|
+
`--no-train-camera` and `--no-test-camera` drop one split. The same
|
|
127
|
+
`--use-colmap` switch writes images to `frame*/input` and runs COLMAP.
|
|
128
|
+
`--start-number` uses the same 1-based output numbering; source file
|
|
129
|
+
`000000.jpg` is frame `1`.
|
|
@@ -6,6 +6,8 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
|
|
|
6
6
|
|
|
7
7
|
- **Neural 3D Video Dataset**: scenes with `poses_bounds.npy` and one `mp4`
|
|
8
8
|
file per camera. See `nvs2colmap/n3dv/README.md`.
|
|
9
|
+
- **Dynamic 3D Gaussians**: scenes with `train_meta.json`, `test_meta.json`,
|
|
10
|
+
images in `ims/`, and masks in `seg/`. See `nvs2colmap/dynamic3dgs/README.md`.
|
|
9
11
|
|
|
10
12
|
## Supported Datasets
|
|
11
13
|
|
|
@@ -20,6 +22,10 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
|
|
|
20
22
|
[liuyubian/Robo360](https://huggingface.co/datasets/liuyubian/Robo360),
|
|
21
23
|
paper
|
|
22
24
|
[Robo360: A 3D Omnispective Multi-Material Robotic Manipulation Dataset](https://arxiv.org/abs/2312.06686).
|
|
25
|
+
- **Dynamic 3D Gaussians**: dataset
|
|
26
|
+
[JonathonLuiten/Dynamic3DGaussians](https://github.com/JonathonLuiten/Dynamic3DGaussians),
|
|
27
|
+
paper
|
|
28
|
+
[Dynamic 3D Gaussians: Tracking by Persistent Dynamic View Synthesis](https://arxiv.org/abs/2308.09713).
|
|
23
29
|
|
|
24
30
|
## Quick Start
|
|
25
31
|
|
|
@@ -60,18 +66,36 @@ Run the full COLMAP pipeline for each frame:
|
|
|
60
66
|
```bash
|
|
61
67
|
python -m nvs2colmap.n3dv \
|
|
62
68
|
--path data/Robo360/xarm6_gold_rope_in_basket_2 \
|
|
63
|
-
--ffmpeg
|
|
64
|
-
--ffprobe
|
|
69
|
+
--ffmpeg ffmpeg \
|
|
70
|
+
--ffprobe ffprobe \
|
|
65
71
|
--video-extension MP4 \
|
|
66
72
|
--n-frames 1 \
|
|
67
73
|
--use-colmap \
|
|
68
|
-
--colmap-executable
|
|
74
|
+
--colmap-executable colmap \
|
|
69
75
|
--colmap-use-gpu 1
|
|
70
76
|
```
|
|
71
77
|
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
+
Link a Dynamic 3D Gaussians scene, including its masks, and write per-frame
|
|
79
|
+
COLMAP text models:
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
python -m nvs2colmap.dynamic3dgs \
|
|
83
|
+
--path data/basketball \
|
|
84
|
+
--n-frames 150
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
For Neural 3D Video scenes, decoded frames are written to `frame*/images` by
|
|
88
|
+
default, and the command also writes `frame*/sparse/0` text models. With
|
|
89
|
+
`--use-colmap`, decoded frames are written to `frame*/input`, and each frame
|
|
90
|
+
additionally gets the standard COLMAP outputs such as `distorted/`, `images/`,
|
|
91
|
+
`sparse/`, and `stereo/`. When `--start-number N` is provided, decoding starts
|
|
92
|
+
from source video frame `N`, and the generated folders/images are also numbered
|
|
93
|
+
from `N`.
|
|
94
|
+
|
|
95
|
+
For Dynamic 3D Gaussians scenes, images from both `train_meta.json` and
|
|
96
|
+
`test_meta.json` are hardlinked into `frame*/images`, and included masks are
|
|
97
|
+
hardlinked into `frame*/image_masks` (`cam01.jpg` pairs with `cam01.jpg.png`).
|
|
98
|
+
`--no-train-camera` and `--no-test-camera` drop one split. The same
|
|
99
|
+
`--use-colmap` switch writes images to `frame*/input` and runs COLMAP.
|
|
100
|
+
`--start-number` uses the same 1-based output numbering; source file
|
|
101
|
+
`000000.jpg` is frame `1`.
|
|
@@ -41,7 +41,6 @@ def run_colmap(
|
|
|
41
41
|
use_gpu: str = "1",
|
|
42
42
|
) -> None:
|
|
43
43
|
folder = Path(folder)
|
|
44
|
-
colmap_executable = os.path.abspath(colmap_executable)
|
|
45
44
|
colmap_cameras, colmap_images = build_colmap_records(cameras, image_extension)
|
|
46
45
|
|
|
47
46
|
if feature_extractor(str(folder), use_gpu=use_gpu, colmap_executable=colmap_executable) != 0:
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""Extract Dynamic 3D Gaussians scenes into per-frame COLMAP folders."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from nvs2colmap.colmap import run_video_colmap
|
|
9
|
+
from nvs2colmap.write_model import write_video_colmap_text_model
|
|
10
|
+
|
|
11
|
+
from .camera_meta import read_camera_meta
|
|
12
|
+
from .link_frames import count_frame_dirs, link_frames
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def parse_args() -> argparse.Namespace:
|
|
16
|
+
parser = argparse.ArgumentParser(
|
|
17
|
+
description=(
|
|
18
|
+
"Extract a Dynamic 3D Gaussians scene and convert train_meta.json "
|
|
19
|
+
"and test_meta.json to per-frame COLMAP text models."
|
|
20
|
+
)
|
|
21
|
+
)
|
|
22
|
+
parser.add_argument(
|
|
23
|
+
"--path",
|
|
24
|
+
type=Path,
|
|
25
|
+
required=True,
|
|
26
|
+
help="Scene directory containing train_meta.json, test_meta.json, ims/, and seg/.",
|
|
27
|
+
)
|
|
28
|
+
parser.add_argument(
|
|
29
|
+
"--n-frames",
|
|
30
|
+
type=int,
|
|
31
|
+
help="Number of frames to extract and write. Defaults to the frames listed in the camera metadata.",
|
|
32
|
+
)
|
|
33
|
+
parser.add_argument(
|
|
34
|
+
"--start-number",
|
|
35
|
+
type=int,
|
|
36
|
+
default=1,
|
|
37
|
+
help="1-based source frame to start extracting from; output frame folders use the same starting number.",
|
|
38
|
+
)
|
|
39
|
+
parser.add_argument(
|
|
40
|
+
"--no-train-camera",
|
|
41
|
+
action="store_true",
|
|
42
|
+
help="Do not extract cameras listed in train_meta.json.",
|
|
43
|
+
)
|
|
44
|
+
parser.add_argument(
|
|
45
|
+
"--no-test-camera",
|
|
46
|
+
action="store_true",
|
|
47
|
+
help="Do not extract cameras listed in test_meta.json.",
|
|
48
|
+
)
|
|
49
|
+
parser.add_argument(
|
|
50
|
+
"--skip-frame-linking",
|
|
51
|
+
action="store_true",
|
|
52
|
+
help="Only convert camera metadata for existing frame folders.",
|
|
53
|
+
)
|
|
54
|
+
parser.add_argument(
|
|
55
|
+
"--use-colmap",
|
|
56
|
+
action="store_true",
|
|
57
|
+
help=(
|
|
58
|
+
"Run COLMAP feature extraction, matching, triangulation, mapping, "
|
|
59
|
+
"and undistortion instead of only writing sparse/0 text models."
|
|
60
|
+
),
|
|
61
|
+
)
|
|
62
|
+
parser.add_argument(
|
|
63
|
+
"--colmap-executable",
|
|
64
|
+
default="colmap",
|
|
65
|
+
help="COLMAP executable used when --use-colmap is set.",
|
|
66
|
+
)
|
|
67
|
+
parser.add_argument(
|
|
68
|
+
"--colmap-use-gpu",
|
|
69
|
+
dest="colmap_use_gpu",
|
|
70
|
+
default="1",
|
|
71
|
+
help="Whether COLMAP SIFT extraction/matching should use GPU when --use-colmap is set.",
|
|
72
|
+
)
|
|
73
|
+
args = parser.parse_args()
|
|
74
|
+
if args.no_train_camera and args.no_test_camera:
|
|
75
|
+
parser.error("Cannot set both --no-train-camera and --no-test-camera.")
|
|
76
|
+
return args
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def main() -> None:
|
|
80
|
+
args = parse_args()
|
|
81
|
+
folder = args.path.resolve()
|
|
82
|
+
|
|
83
|
+
cameras, available_frames = read_camera_meta(
|
|
84
|
+
folder,
|
|
85
|
+
include_train=not args.no_train_camera,
|
|
86
|
+
include_test=not args.no_test_camera,
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
n_frames = args.n_frames
|
|
90
|
+
frame_output_pattern = folder / "frame%d"
|
|
91
|
+
image_dir_name = "input" if args.use_colmap else "images"
|
|
92
|
+
if not args.skip_frame_linking:
|
|
93
|
+
if n_frames is None:
|
|
94
|
+
n_frames = available_frames - args.start_number + 1
|
|
95
|
+
link_frames(
|
|
96
|
+
folder=folder,
|
|
97
|
+
frame_pattern=frame_output_pattern,
|
|
98
|
+
cameras=cameras,
|
|
99
|
+
n_frames=n_frames,
|
|
100
|
+
start_number=args.start_number,
|
|
101
|
+
image_dirname=image_dir_name,
|
|
102
|
+
)
|
|
103
|
+
elif n_frames is None:
|
|
104
|
+
n_frames = count_frame_dirs(frame_output_pattern, start_number=args.start_number)
|
|
105
|
+
|
|
106
|
+
colmap_cameras = [camera.camera for camera in cameras]
|
|
107
|
+
if not args.use_colmap:
|
|
108
|
+
write_video_colmap_text_model(
|
|
109
|
+
output_pattern=frame_output_pattern / "sparse" / "0",
|
|
110
|
+
cameras=colmap_cameras,
|
|
111
|
+
n_frames=n_frames,
|
|
112
|
+
start_number=args.start_number,
|
|
113
|
+
image_extension="",
|
|
114
|
+
)
|
|
115
|
+
else:
|
|
116
|
+
run_video_colmap(
|
|
117
|
+
output_pattern=frame_output_pattern,
|
|
118
|
+
cameras=colmap_cameras,
|
|
119
|
+
n_frames=n_frames,
|
|
120
|
+
start_number=args.start_number,
|
|
121
|
+
image_extension="",
|
|
122
|
+
colmap_executable=args.colmap_executable,
|
|
123
|
+
use_gpu=args.colmap_use_gpu,
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
print(f"Done: {folder}")
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
if __name__ == "__main__":
|
|
130
|
+
main()
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""Read Dynamic 3D Gaussians train and test camera metadata."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import torch
|
|
10
|
+
|
|
11
|
+
from nvs2colmap.utils import matrix_to_quaternion
|
|
12
|
+
from nvs2colmap.write_model import CameraModel
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class ParsedCameraMeta:
|
|
17
|
+
"""One static camera and the image filenames for each frame."""
|
|
18
|
+
|
|
19
|
+
camera: CameraModel
|
|
20
|
+
filenames: list[str]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def read_meta(folder: Path, meta_name: str) -> tuple[list[ParsedCameraMeta], int]:
|
|
24
|
+
folder = Path(folder)
|
|
25
|
+
meta_path = folder / meta_name
|
|
26
|
+
with meta_path.open() as f:
|
|
27
|
+
camera_meta = json.load(f)
|
|
28
|
+
|
|
29
|
+
n_frames = len(camera_meta["fn"])
|
|
30
|
+
if n_frames == 0:
|
|
31
|
+
raise ValueError(f"No frames found in {meta_path}")
|
|
32
|
+
for key in ("k", "w2c", "cam_id"):
|
|
33
|
+
if any(camera_meta[key][frame_index] != camera_meta[key][0] for frame_index in range(n_frames)):
|
|
34
|
+
raise ValueError(f"{meta_name} {key} changes across frames.")
|
|
35
|
+
|
|
36
|
+
width = int(camera_meta["w"])
|
|
37
|
+
height = int(camera_meta["h"])
|
|
38
|
+
cameras = []
|
|
39
|
+
seen_ids = set()
|
|
40
|
+
for camera_index, (k, w2c, camera_id) in enumerate(
|
|
41
|
+
zip(camera_meta["k"][0], camera_meta["w2c"][0], camera_meta["cam_id"][0])
|
|
42
|
+
):
|
|
43
|
+
camera_id = int(camera_id)
|
|
44
|
+
if camera_id in seen_ids:
|
|
45
|
+
raise ValueError(f"{meta_name} lists camera {camera_id} more than once.")
|
|
46
|
+
seen_ids.add(camera_id)
|
|
47
|
+
filenames = [camera_meta["fn"][frame_index][camera_index] for frame_index in range(n_frames)]
|
|
48
|
+
w2c_tensor = torch.tensor(w2c, dtype=torch.float64)
|
|
49
|
+
quaternion = matrix_to_quaternion(w2c_tensor[:3, :3])
|
|
50
|
+
cameras.append(
|
|
51
|
+
ParsedCameraMeta(
|
|
52
|
+
camera=CameraModel(
|
|
53
|
+
name=f"cam{camera_id:02d}{Path(filenames[0]).suffix}",
|
|
54
|
+
width=width,
|
|
55
|
+
height=height,
|
|
56
|
+
fx=float(k[0][0]),
|
|
57
|
+
fy=float(k[1][1]),
|
|
58
|
+
cx=float(k[0][2]),
|
|
59
|
+
cy=float(k[1][2]),
|
|
60
|
+
qvec=quaternion.detach().cpu().numpy(),
|
|
61
|
+
tvec=w2c_tensor[:3, 3].detach().cpu().numpy(),
|
|
62
|
+
),
|
|
63
|
+
filenames=filenames,
|
|
64
|
+
)
|
|
65
|
+
)
|
|
66
|
+
return cameras, n_frames
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def read_camera_meta(
|
|
70
|
+
folder: Path,
|
|
71
|
+
include_train: bool = True,
|
|
72
|
+
include_test: bool = True,
|
|
73
|
+
) -> tuple[list[ParsedCameraMeta], int]:
|
|
74
|
+
if not include_train and not include_test:
|
|
75
|
+
raise ValueError("At least one of train and test cameras must be included.")
|
|
76
|
+
|
|
77
|
+
n_frames = None
|
|
78
|
+
train_cameras, train_n_frames = [], None
|
|
79
|
+
if include_train:
|
|
80
|
+
train_cameras, train_n_frames = read_meta(folder, "train_meta.json")
|
|
81
|
+
n_frames = train_n_frames
|
|
82
|
+
test_cameras, test_n_frames = [], None
|
|
83
|
+
if include_test:
|
|
84
|
+
test_cameras, test_n_frames = read_meta(folder, "test_meta.json")
|
|
85
|
+
n_frames = test_n_frames
|
|
86
|
+
|
|
87
|
+
if include_train and include_test and train_n_frames != test_n_frames:
|
|
88
|
+
raise ValueError("train_meta.json and test_meta.json list different frame counts.")
|
|
89
|
+
cameras = train_cameras + test_cameras
|
|
90
|
+
|
|
91
|
+
merged: dict[str, ParsedCameraMeta] = {}
|
|
92
|
+
for camera in cameras:
|
|
93
|
+
if camera.camera.name in merged:
|
|
94
|
+
raise ValueError(f"Camera {camera.camera.name} appears in more than one metadata file.")
|
|
95
|
+
merged[camera.camera.name] = camera
|
|
96
|
+
if not merged:
|
|
97
|
+
raise ValueError("No cameras selected.")
|
|
98
|
+
return [merged[name] for name in sorted(merged)], n_frames
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Link Dynamic 3D Gaussians images and masks into per-frame folders."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Sequence
|
|
8
|
+
|
|
9
|
+
from .camera_meta import ParsedCameraMeta
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def count_frame_dirs(output_pattern: Path, start_number: int = 1) -> int:
|
|
13
|
+
output_pattern = str(output_pattern)
|
|
14
|
+
frame = start_number
|
|
15
|
+
while Path(output_pattern % frame).is_dir():
|
|
16
|
+
frame += 1
|
|
17
|
+
n_frames = frame - start_number
|
|
18
|
+
if n_frames == 0:
|
|
19
|
+
raise FileNotFoundError(f"No frame directories found from pattern: {output_pattern}")
|
|
20
|
+
return n_frames
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def link_file(src: Path, dst: Path) -> None:
|
|
24
|
+
if not src.is_file():
|
|
25
|
+
raise FileNotFoundError(f"Missing source file: {src}")
|
|
26
|
+
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
27
|
+
if dst.exists():
|
|
28
|
+
dst.unlink()
|
|
29
|
+
os.link(src, dst)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def link_frames(
|
|
33
|
+
folder: Path,
|
|
34
|
+
frame_pattern: Path,
|
|
35
|
+
cameras: Sequence[ParsedCameraMeta],
|
|
36
|
+
n_frames: int,
|
|
37
|
+
start_number: int = 1,
|
|
38
|
+
image_dirname: str = "images",
|
|
39
|
+
) -> None:
|
|
40
|
+
folder = Path(folder)
|
|
41
|
+
frame_pattern = str(frame_pattern)
|
|
42
|
+
for offset in range(n_frames):
|
|
43
|
+
frame_index = start_number - 1 + offset
|
|
44
|
+
frame_dir = Path(frame_pattern % (start_number + offset))
|
|
45
|
+
image_dir = frame_dir / image_dirname
|
|
46
|
+
mask_dir = frame_dir / "image_masks"
|
|
47
|
+
for camera in cameras:
|
|
48
|
+
filename = camera.filenames[frame_index]
|
|
49
|
+
link_file(folder / "ims" / filename, image_dir / camera.camera.name)
|
|
50
|
+
mask_src = folder / "seg" / Path(filename).with_suffix(".png")
|
|
51
|
+
if mask_src.is_file():
|
|
52
|
+
link_file(mask_src, mask_dir / f"{camera.camera.name}.png")
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""Extract one or more videos into per-frame image folders."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from nvs2colmap.utils import extract_video_frames_parallel
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def extract_videos(
|
|
12
|
+
video_paths: list[Path],
|
|
13
|
+
output_pattern: Path,
|
|
14
|
+
start_number: int = 1,
|
|
15
|
+
n_frames: int | None = None,
|
|
16
|
+
ffmpeg_executable: str = "ffmpeg",
|
|
17
|
+
ffprobe_executable: str = "ffprobe",
|
|
18
|
+
ffmpeg_processes: int = 1,
|
|
19
|
+
image_extension: str = ".png",
|
|
20
|
+
) -> int:
|
|
21
|
+
if not image_extension.startswith("."):
|
|
22
|
+
image_extension = f".{image_extension}"
|
|
23
|
+
|
|
24
|
+
jobs = []
|
|
25
|
+
for index, video_path in enumerate(video_paths, start=1):
|
|
26
|
+
if not video_path.is_file():
|
|
27
|
+
raise FileNotFoundError(f"Missing video file: {video_path}")
|
|
28
|
+
jobs.append((video_path, str(output_pattern / f"{index:04d}{image_extension}")))
|
|
29
|
+
|
|
30
|
+
if not jobs:
|
|
31
|
+
raise ValueError("At least one video path is required.")
|
|
32
|
+
|
|
33
|
+
extracted_frame_counts = extract_video_frames_parallel(
|
|
34
|
+
jobs,
|
|
35
|
+
start_number=start_number,
|
|
36
|
+
n_frames=n_frames,
|
|
37
|
+
ffmpeg_executable=ffmpeg_executable,
|
|
38
|
+
ffprobe_executable=ffprobe_executable,
|
|
39
|
+
process_count=ffmpeg_processes,
|
|
40
|
+
)
|
|
41
|
+
return max(extracted_frame_counts)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def parse_args() -> argparse.Namespace:
|
|
45
|
+
parser = argparse.ArgumentParser(
|
|
46
|
+
description=(
|
|
47
|
+
"Extract videos into frame folders. For input videos v1.mp4 v2.mp4 "
|
|
48
|
+
"and --output-pattern 'frame%d/images', extracted images are named "
|
|
49
|
+
"frame1/images/0001.png, frame1/images/0002.png, etc."
|
|
50
|
+
)
|
|
51
|
+
)
|
|
52
|
+
parser.add_argument(
|
|
53
|
+
"videos",
|
|
54
|
+
type=Path,
|
|
55
|
+
nargs="+",
|
|
56
|
+
help="Video files to extract. Image names follow this input order.",
|
|
57
|
+
)
|
|
58
|
+
parser.add_argument(
|
|
59
|
+
"--output-pattern",
|
|
60
|
+
type=Path,
|
|
61
|
+
required=True,
|
|
62
|
+
help="Per-frame output directory pattern, for example 'frame%%d/images'.",
|
|
63
|
+
)
|
|
64
|
+
parser.add_argument(
|
|
65
|
+
"--n-frames",
|
|
66
|
+
type=int,
|
|
67
|
+
help="Number of frames to extract. Defaults to each video's available frame count.",
|
|
68
|
+
)
|
|
69
|
+
parser.add_argument(
|
|
70
|
+
"--start-number",
|
|
71
|
+
type=int,
|
|
72
|
+
default=1,
|
|
73
|
+
help="1-based source frame to start extracting from; output frame folders use the same starting number.",
|
|
74
|
+
)
|
|
75
|
+
parser.add_argument(
|
|
76
|
+
"--ffmpeg",
|
|
77
|
+
dest="ffmpeg_executable",
|
|
78
|
+
default="ffmpeg",
|
|
79
|
+
help="ffmpeg executable.",
|
|
80
|
+
)
|
|
81
|
+
parser.add_argument(
|
|
82
|
+
"--ffprobe",
|
|
83
|
+
dest="ffprobe_executable",
|
|
84
|
+
default="ffprobe",
|
|
85
|
+
help="ffprobe executable.",
|
|
86
|
+
)
|
|
87
|
+
parser.add_argument(
|
|
88
|
+
"--ffmpeg-processes",
|
|
89
|
+
type=int,
|
|
90
|
+
default=1,
|
|
91
|
+
help="Number of ffmpeg processes to run in parallel.",
|
|
92
|
+
)
|
|
93
|
+
parser.add_argument(
|
|
94
|
+
"--image-extension",
|
|
95
|
+
default=".png",
|
|
96
|
+
help="Image extension written by ffmpeg.",
|
|
97
|
+
)
|
|
98
|
+
return parser.parse_args()
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def main() -> None:
|
|
102
|
+
args = parse_args()
|
|
103
|
+
n_frames = extract_videos(
|
|
104
|
+
video_paths=args.videos,
|
|
105
|
+
output_pattern=args.output_pattern,
|
|
106
|
+
start_number=args.start_number,
|
|
107
|
+
n_frames=args.n_frames,
|
|
108
|
+
ffmpeg_executable=args.ffmpeg_executable,
|
|
109
|
+
ffprobe_executable=args.ffprobe_executable,
|
|
110
|
+
ffmpeg_processes=args.ffmpeg_processes,
|
|
111
|
+
image_extension=args.image_extension,
|
|
112
|
+
)
|
|
113
|
+
print(f"Done: extracted up to {n_frames} frames per video.")
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
if __name__ == "__main__":
|
|
117
|
+
main()
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: nvs2colmap
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Utilities for converting novel view synthesis datasets to COLMAP format.
|
|
5
5
|
Author-email: Howard Yin <yindaheng98@gmail.com>
|
|
6
6
|
Maintainer-email: Howard Yin <yindaheng98@gmail.com>
|
|
@@ -34,6 +34,8 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
|
|
|
34
34
|
|
|
35
35
|
- **Neural 3D Video Dataset**: scenes with `poses_bounds.npy` and one `mp4`
|
|
36
36
|
file per camera. See `nvs2colmap/n3dv/README.md`.
|
|
37
|
+
- **Dynamic 3D Gaussians**: scenes with `train_meta.json`, `test_meta.json`,
|
|
38
|
+
images in `ims/`, and masks in `seg/`. See `nvs2colmap/dynamic3dgs/README.md`.
|
|
37
39
|
|
|
38
40
|
## Supported Datasets
|
|
39
41
|
|
|
@@ -48,6 +50,10 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
|
|
|
48
50
|
[liuyubian/Robo360](https://huggingface.co/datasets/liuyubian/Robo360),
|
|
49
51
|
paper
|
|
50
52
|
[Robo360: A 3D Omnispective Multi-Material Robotic Manipulation Dataset](https://arxiv.org/abs/2312.06686).
|
|
53
|
+
- **Dynamic 3D Gaussians**: dataset
|
|
54
|
+
[JonathonLuiten/Dynamic3DGaussians](https://github.com/JonathonLuiten/Dynamic3DGaussians),
|
|
55
|
+
paper
|
|
56
|
+
[Dynamic 3D Gaussians: Tracking by Persistent Dynamic View Synthesis](https://arxiv.org/abs/2308.09713).
|
|
51
57
|
|
|
52
58
|
## Quick Start
|
|
53
59
|
|
|
@@ -88,18 +94,36 @@ Run the full COLMAP pipeline for each frame:
|
|
|
88
94
|
```bash
|
|
89
95
|
python -m nvs2colmap.n3dv \
|
|
90
96
|
--path data/Robo360/xarm6_gold_rope_in_basket_2 \
|
|
91
|
-
--ffmpeg
|
|
92
|
-
--ffprobe
|
|
97
|
+
--ffmpeg ffmpeg \
|
|
98
|
+
--ffprobe ffprobe \
|
|
93
99
|
--video-extension MP4 \
|
|
94
100
|
--n-frames 1 \
|
|
95
101
|
--use-colmap \
|
|
96
|
-
--colmap-executable
|
|
102
|
+
--colmap-executable colmap \
|
|
97
103
|
--colmap-use-gpu 1
|
|
98
104
|
```
|
|
99
105
|
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
+
Link a Dynamic 3D Gaussians scene, including its masks, and write per-frame
|
|
107
|
+
COLMAP text models:
|
|
108
|
+
|
|
109
|
+
```bash
|
|
110
|
+
python -m nvs2colmap.dynamic3dgs \
|
|
111
|
+
--path data/basketball \
|
|
112
|
+
--n-frames 150
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
For Neural 3D Video scenes, decoded frames are written to `frame*/images` by
|
|
116
|
+
default, and the command also writes `frame*/sparse/0` text models. With
|
|
117
|
+
`--use-colmap`, decoded frames are written to `frame*/input`, and each frame
|
|
118
|
+
additionally gets the standard COLMAP outputs such as `distorted/`, `images/`,
|
|
119
|
+
`sparse/`, and `stereo/`. When `--start-number N` is provided, decoding starts
|
|
120
|
+
from source video frame `N`, and the generated folders/images are also numbered
|
|
121
|
+
from `N`.
|
|
122
|
+
|
|
123
|
+
For Dynamic 3D Gaussians scenes, images from both `train_meta.json` and
|
|
124
|
+
`test_meta.json` are hardlinked into `frame*/images`, and included masks are
|
|
125
|
+
hardlinked into `frame*/image_masks` (`cam01.jpg` pairs with `cam01.jpg.png`).
|
|
126
|
+
`--no-train-camera` and `--no-test-camera` drop one split. The same
|
|
127
|
+
`--use-colmap` switch writes images to `frame*/input` and runs COLMAP.
|
|
128
|
+
`--start-number` uses the same 1-based output numbering; source file
|
|
129
|
+
`000000.jpg` is frame `1`.
|
|
@@ -3,8 +3,11 @@ README.md
|
|
|
3
3
|
pyproject.toml
|
|
4
4
|
./nvs2colmap/__init__.py
|
|
5
5
|
./nvs2colmap/colmap.py
|
|
6
|
+
./nvs2colmap/extract_videos.py
|
|
6
7
|
./nvs2colmap/write_model.py
|
|
7
|
-
./nvs2colmap/dynamic3dgs/
|
|
8
|
+
./nvs2colmap/dynamic3dgs/__main__.py
|
|
9
|
+
./nvs2colmap/dynamic3dgs/camera_meta.py
|
|
10
|
+
./nvs2colmap/dynamic3dgs/link_frames.py
|
|
8
11
|
./nvs2colmap/n3dv/__main__.py
|
|
9
12
|
./nvs2colmap/n3dv/extract_videos.py
|
|
10
13
|
./nvs2colmap/n3dv/poses_bounds.py
|
|
@@ -15,6 +18,7 @@ pyproject.toml
|
|
|
15
18
|
./nvs2colmap/utils/rotation.py
|
|
16
19
|
nvs2colmap/__init__.py
|
|
17
20
|
nvs2colmap/colmap.py
|
|
21
|
+
nvs2colmap/extract_videos.py
|
|
18
22
|
nvs2colmap/write_model.py
|
|
19
23
|
nvs2colmap.egg-info/PKG-INFO
|
|
20
24
|
nvs2colmap.egg-info/SOURCES.txt
|
|
@@ -22,7 +26,9 @@ nvs2colmap.egg-info/dependency_links.txt
|
|
|
22
26
|
nvs2colmap.egg-info/entry_points.txt
|
|
23
27
|
nvs2colmap.egg-info/requires.txt
|
|
24
28
|
nvs2colmap.egg-info/top_level.txt
|
|
25
|
-
nvs2colmap/dynamic3dgs/
|
|
29
|
+
nvs2colmap/dynamic3dgs/__main__.py
|
|
30
|
+
nvs2colmap/dynamic3dgs/camera_meta.py
|
|
31
|
+
nvs2colmap/dynamic3dgs/link_frames.py
|
|
26
32
|
nvs2colmap/n3dv/__main__.py
|
|
27
33
|
nvs2colmap/n3dv/extract_videos.py
|
|
28
34
|
nvs2colmap/n3dv/poses_bounds.py
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "nvs2colmap"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.3.0"
|
|
8
8
|
description = "Utilities for converting novel view synthesis datasets to COLMAP format."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
authors = [{ name = "Howard Yin", email = "yindaheng98@gmail.com" }]
|
|
@@ -41,7 +41,9 @@ Repository = "https://github.com/yindaheng98/NVS2COLMAP"
|
|
|
41
41
|
Issues = "https://github.com/yindaheng98/NVS2COLMAP/issues"
|
|
42
42
|
|
|
43
43
|
[project.scripts]
|
|
44
|
+
nvs2colmap-extract-videos = "nvs2colmap.extract_videos:main"
|
|
44
45
|
nvs2colmap-n3dv = "nvs2colmap.n3dv.__main__:main"
|
|
46
|
+
nvs2colmap-dynamic3dgs = "nvs2colmap.dynamic3dgs.__main__:main"
|
|
45
47
|
|
|
46
48
|
[tool.setuptools]
|
|
47
49
|
packages = { find = { include = ["nvs2colmap*"] } }
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|