nvs2colmap 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/PKG-INFO +34 -10
  2. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/README.md +33 -9
  3. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/colmap.py +0 -1
  4. nvs2colmap-0.3.0/nvs2colmap/dynamic3dgs/__main__.py +130 -0
  5. nvs2colmap-0.3.0/nvs2colmap/dynamic3dgs/camera_meta.py +98 -0
  6. nvs2colmap-0.3.0/nvs2colmap/dynamic3dgs/link_frames.py +52 -0
  7. nvs2colmap-0.3.0/nvs2colmap/extract_videos.py +117 -0
  8. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap.egg-info/PKG-INFO +34 -10
  9. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap.egg-info/SOURCES.txt +8 -2
  10. nvs2colmap-0.3.0/nvs2colmap.egg-info/entry_points.txt +4 -0
  11. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/pyproject.toml +3 -1
  12. nvs2colmap-0.2.0/nvs2colmap/stnerf/__init__.py +0 -0
  13. nvs2colmap-0.2.0/nvs2colmap.egg-info/entry_points.txt +0 -2
  14. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/LICENSE +0 -0
  15. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/__init__.py +0 -0
  16. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/n3dv/__main__.py +0 -0
  17. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/n3dv/extract_videos.py +0 -0
  18. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/n3dv/poses_bounds.py +0 -0
  19. {nvs2colmap-0.2.0/nvs2colmap/dynamic3dgs → nvs2colmap-0.3.0/nvs2colmap/stnerf}/__init__.py +0 -0
  20. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/utils/__init__.py +0 -0
  21. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/utils/colmap.py +0 -0
  22. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/utils/ffmpeg.py +0 -0
  23. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/utils/rotation.py +0 -0
  24. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap/write_model.py +0 -0
  25. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap.egg-info/dependency_links.txt +0 -0
  26. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap.egg-info/requires.txt +0 -0
  27. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/nvs2colmap.egg-info/top_level.txt +0 -0
  28. {nvs2colmap-0.2.0 → nvs2colmap-0.3.0}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: nvs2colmap
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Utilities for converting novel view synthesis datasets to COLMAP format.
5
5
  Author-email: Howard Yin <yindaheng98@gmail.com>
6
6
  Maintainer-email: Howard Yin <yindaheng98@gmail.com>
@@ -34,6 +34,8 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
34
34
 
35
35
  - **Neural 3D Video Dataset**: scenes with `poses_bounds.npy` and one `mp4`
36
36
  file per camera. See `nvs2colmap/n3dv/README.md`.
37
+ - **Dynamic 3D Gaussians**: scenes with `train_meta.json`, `test_meta.json`,
38
+ images in `ims/`, and masks in `seg/`. See `nvs2colmap/dynamic3dgs/README.md`.
37
39
 
38
40
  ## Supported Datasets
39
41
 
@@ -48,6 +50,10 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
48
50
  [liuyubian/Robo360](https://huggingface.co/datasets/liuyubian/Robo360),
49
51
  paper
50
52
  [Robo360: A 3D Omnispective Multi-Material Robotic Manipulation Dataset](https://arxiv.org/abs/2312.06686).
53
+ - **Dynamic 3D Gaussians**: dataset
54
+ [JonathonLuiten/Dynamic3DGaussians](https://github.com/JonathonLuiten/Dynamic3DGaussians),
55
+ paper
56
+ [Dynamic 3D Gaussians: Tracking by Persistent Dynamic View Synthesis](https://arxiv.org/abs/2308.09713).
51
57
 
52
58
  ## Quick Start
53
59
 
@@ -88,18 +94,36 @@ Run the full COLMAP pipeline for each frame:
88
94
  ```bash
89
95
  python -m nvs2colmap.n3dv \
90
96
  --path data/Robo360/xarm6_gold_rope_in_basket_2 \
91
- --ffmpeg D:/MyPrograms/ffmpeg.exe \
92
- --ffprobe D:/MyPrograms/ffprobe.exe \
97
+ --ffmpeg ffmpeg \
98
+ --ffprobe ffprobe \
93
99
  --video-extension MP4 \
94
100
  --n-frames 1 \
95
101
  --use-colmap \
96
- --colmap-executable data/colmap/COLMAP.bat \
102
+ --colmap-executable colmap \
97
103
  --colmap-use-gpu 1
98
104
  ```
99
105
 
100
- By default, decoded frames are written to `frame*/images`, and the command also
101
- writes `frame*/sparse/0` text models. With `--use-colmap`, decoded frames are
102
- written to `frame*/input`, and each frame additionally gets the standard
103
- COLMAP outputs such as `distorted/`, `images/`, `sparse/`, and `stereo/`. When
104
- `--start-number N` is provided, decoding starts from source video frame `N`, and
105
- the generated folders/images are also numbered from `N`.
106
+ Link a Dynamic 3D Gaussians scene, including its masks, and write per-frame
107
+ COLMAP text models:
108
+
109
+ ```bash
110
+ python -m nvs2colmap.dynamic3dgs \
111
+ --path data/basketball \
112
+ --n-frames 150
113
+ ```
114
+
115
+ For Neural 3D Video scenes, decoded frames are written to `frame*/images` by
116
+ default, and the command also writes `frame*/sparse/0` text models. With
117
+ `--use-colmap`, decoded frames are written to `frame*/input`, and each frame
118
+ additionally gets the standard COLMAP outputs such as `distorted/`, `images/`,
119
+ `sparse/`, and `stereo/`. When `--start-number N` is provided, decoding starts
120
+ from source video frame `N`, and the generated folders/images are also numbered
121
+ from `N`.
122
+
123
+ For Dynamic 3D Gaussians scenes, images from both `train_meta.json` and
124
+ `test_meta.json` are hardlinked into `frame*/images`, and included masks are
125
+ hardlinked into `frame*/image_masks` (`cam01.jpg` pairs with `cam01.jpg.png`).
126
+ `--no-train-camera` and `--no-test-camera` drop one split. The same
127
+ `--use-colmap` switch writes images to `frame*/input` and runs COLMAP.
128
+ `--start-number` uses the same 1-based output numbering; source file
129
+ `000000.jpg` is frame `1`.
@@ -6,6 +6,8 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
6
6
 
7
7
  - **Neural 3D Video Dataset**: scenes with `poses_bounds.npy` and one `mp4`
8
8
  file per camera. See `nvs2colmap/n3dv/README.md`.
9
+ - **Dynamic 3D Gaussians**: scenes with `train_meta.json`, `test_meta.json`,
10
+ images in `ims/`, and masks in `seg/`. See `nvs2colmap/dynamic3dgs/README.md`.
9
11
 
10
12
  ## Supported Datasets
11
13
 
@@ -20,6 +22,10 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
20
22
  [liuyubian/Robo360](https://huggingface.co/datasets/liuyubian/Robo360),
21
23
  paper
22
24
  [Robo360: A 3D Omnispective Multi-Material Robotic Manipulation Dataset](https://arxiv.org/abs/2312.06686).
25
+ - **Dynamic 3D Gaussians**: dataset
26
+ [JonathonLuiten/Dynamic3DGaussians](https://github.com/JonathonLuiten/Dynamic3DGaussians),
27
+ paper
28
+ [Dynamic 3D Gaussians: Tracking by Persistent Dynamic View Synthesis](https://arxiv.org/abs/2308.09713).
23
29
 
24
30
  ## Quick Start
25
31
 
@@ -60,18 +66,36 @@ Run the full COLMAP pipeline for each frame:
60
66
  ```bash
61
67
  python -m nvs2colmap.n3dv \
62
68
  --path data/Robo360/xarm6_gold_rope_in_basket_2 \
63
- --ffmpeg D:/MyPrograms/ffmpeg.exe \
64
- --ffprobe D:/MyPrograms/ffprobe.exe \
69
+ --ffmpeg ffmpeg \
70
+ --ffprobe ffprobe \
65
71
  --video-extension MP4 \
66
72
  --n-frames 1 \
67
73
  --use-colmap \
68
- --colmap-executable data/colmap/COLMAP.bat \
74
+ --colmap-executable colmap \
69
75
  --colmap-use-gpu 1
70
76
  ```
71
77
 
72
- By default, decoded frames are written to `frame*/images`, and the command also
73
- writes `frame*/sparse/0` text models. With `--use-colmap`, decoded frames are
74
- written to `frame*/input`, and each frame additionally gets the standard
75
- COLMAP outputs such as `distorted/`, `images/`, `sparse/`, and `stereo/`. When
76
- `--start-number N` is provided, decoding starts from source video frame `N`, and
77
- the generated folders/images are also numbered from `N`.
78
+ Link a Dynamic 3D Gaussians scene, including its masks, and write per-frame
79
+ COLMAP text models:
80
+
81
+ ```bash
82
+ python -m nvs2colmap.dynamic3dgs \
83
+ --path data/basketball \
84
+ --n-frames 150
85
+ ```
86
+
87
+ For Neural 3D Video scenes, decoded frames are written to `frame*/images` by
88
+ default, and the command also writes `frame*/sparse/0` text models. With
89
+ `--use-colmap`, decoded frames are written to `frame*/input`, and each frame
90
+ additionally gets the standard COLMAP outputs such as `distorted/`, `images/`,
91
+ `sparse/`, and `stereo/`. When `--start-number N` is provided, decoding starts
92
+ from source video frame `N`, and the generated folders/images are also numbered
93
+ from `N`.
94
+
95
+ For Dynamic 3D Gaussians scenes, images from both `train_meta.json` and
96
+ `test_meta.json` are hardlinked into `frame*/images`, and included masks are
97
+ hardlinked into `frame*/image_masks` (`cam01.jpg` pairs with `cam01.jpg.png`).
98
+ `--no-train-camera` and `--no-test-camera` drop one split. The same
99
+ `--use-colmap` switch writes images to `frame*/input` and runs COLMAP.
100
+ `--start-number` uses the same 1-based output numbering; source file
101
+ `000000.jpg` is frame `1`.
@@ -41,7 +41,6 @@ def run_colmap(
41
41
  use_gpu: str = "1",
42
42
  ) -> None:
43
43
  folder = Path(folder)
44
- colmap_executable = os.path.abspath(colmap_executable)
45
44
  colmap_cameras, colmap_images = build_colmap_records(cameras, image_extension)
46
45
 
47
46
  if feature_extractor(str(folder), use_gpu=use_gpu, colmap_executable=colmap_executable) != 0:
@@ -0,0 +1,130 @@
1
+ """Extract Dynamic 3D Gaussians scenes into per-frame COLMAP folders."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ from pathlib import Path
7
+
8
+ from nvs2colmap.colmap import run_video_colmap
9
+ from nvs2colmap.write_model import write_video_colmap_text_model
10
+
11
+ from .camera_meta import read_camera_meta
12
+ from .link_frames import count_frame_dirs, link_frames
13
+
14
+
15
+ def parse_args() -> argparse.Namespace:
16
+ parser = argparse.ArgumentParser(
17
+ description=(
18
+ "Extract a Dynamic 3D Gaussians scene and convert train_meta.json "
19
+ "and test_meta.json to per-frame COLMAP text models."
20
+ )
21
+ )
22
+ parser.add_argument(
23
+ "--path",
24
+ type=Path,
25
+ required=True,
26
+ help="Scene directory containing train_meta.json, test_meta.json, ims/, and seg/.",
27
+ )
28
+ parser.add_argument(
29
+ "--n-frames",
30
+ type=int,
31
+ help="Number of frames to extract and write. Defaults to the frames listed in the camera metadata.",
32
+ )
33
+ parser.add_argument(
34
+ "--start-number",
35
+ type=int,
36
+ default=1,
37
+ help="1-based source frame to start extracting from; output frame folders use the same starting number.",
38
+ )
39
+ parser.add_argument(
40
+ "--no-train-camera",
41
+ action="store_true",
42
+ help="Do not extract cameras listed in train_meta.json.",
43
+ )
44
+ parser.add_argument(
45
+ "--no-test-camera",
46
+ action="store_true",
47
+ help="Do not extract cameras listed in test_meta.json.",
48
+ )
49
+ parser.add_argument(
50
+ "--skip-frame-linking",
51
+ action="store_true",
52
+ help="Only convert camera metadata for existing frame folders.",
53
+ )
54
+ parser.add_argument(
55
+ "--use-colmap",
56
+ action="store_true",
57
+ help=(
58
+ "Run COLMAP feature extraction, matching, triangulation, mapping, "
59
+ "and undistortion instead of only writing sparse/0 text models."
60
+ ),
61
+ )
62
+ parser.add_argument(
63
+ "--colmap-executable",
64
+ default="colmap",
65
+ help="COLMAP executable used when --use-colmap is set.",
66
+ )
67
+ parser.add_argument(
68
+ "--colmap-use-gpu",
69
+ dest="colmap_use_gpu",
70
+ default="1",
71
+ help="Whether COLMAP SIFT extraction/matching should use GPU when --use-colmap is set.",
72
+ )
73
+ args = parser.parse_args()
74
+ if args.no_train_camera and args.no_test_camera:
75
+ parser.error("Cannot set both --no-train-camera and --no-test-camera.")
76
+ return args
77
+
78
+
79
+ def main() -> None:
80
+ args = parse_args()
81
+ folder = args.path.resolve()
82
+
83
+ cameras, available_frames = read_camera_meta(
84
+ folder,
85
+ include_train=not args.no_train_camera,
86
+ include_test=not args.no_test_camera,
87
+ )
88
+
89
+ n_frames = args.n_frames
90
+ frame_output_pattern = folder / "frame%d"
91
+ image_dir_name = "input" if args.use_colmap else "images"
92
+ if not args.skip_frame_linking:
93
+ if n_frames is None:
94
+ n_frames = available_frames - args.start_number + 1
95
+ link_frames(
96
+ folder=folder,
97
+ frame_pattern=frame_output_pattern,
98
+ cameras=cameras,
99
+ n_frames=n_frames,
100
+ start_number=args.start_number,
101
+ image_dirname=image_dir_name,
102
+ )
103
+ elif n_frames is None:
104
+ n_frames = count_frame_dirs(frame_output_pattern, start_number=args.start_number)
105
+
106
+ colmap_cameras = [camera.camera for camera in cameras]
107
+ if not args.use_colmap:
108
+ write_video_colmap_text_model(
109
+ output_pattern=frame_output_pattern / "sparse" / "0",
110
+ cameras=colmap_cameras,
111
+ n_frames=n_frames,
112
+ start_number=args.start_number,
113
+ image_extension="",
114
+ )
115
+ else:
116
+ run_video_colmap(
117
+ output_pattern=frame_output_pattern,
118
+ cameras=colmap_cameras,
119
+ n_frames=n_frames,
120
+ start_number=args.start_number,
121
+ image_extension="",
122
+ colmap_executable=args.colmap_executable,
123
+ use_gpu=args.colmap_use_gpu,
124
+ )
125
+
126
+ print(f"Done: {folder}")
127
+
128
+
129
+ if __name__ == "__main__":
130
+ main()
@@ -0,0 +1,98 @@
1
+ """Read Dynamic 3D Gaussians train and test camera metadata."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from dataclasses import dataclass
7
+ from pathlib import Path
8
+
9
+ import torch
10
+
11
+ from nvs2colmap.utils import matrix_to_quaternion
12
+ from nvs2colmap.write_model import CameraModel
13
+
14
+
15
+ @dataclass(frozen=True)
16
+ class ParsedCameraMeta:
17
+ """One static camera and the image filenames for each frame."""
18
+
19
+ camera: CameraModel
20
+ filenames: list[str]
21
+
22
+
23
+ def read_meta(folder: Path, meta_name: str) -> tuple[list[ParsedCameraMeta], int]:
24
+ folder = Path(folder)
25
+ meta_path = folder / meta_name
26
+ with meta_path.open() as f:
27
+ camera_meta = json.load(f)
28
+
29
+ n_frames = len(camera_meta["fn"])
30
+ if n_frames == 0:
31
+ raise ValueError(f"No frames found in {meta_path}")
32
+ for key in ("k", "w2c", "cam_id"):
33
+ if any(camera_meta[key][frame_index] != camera_meta[key][0] for frame_index in range(n_frames)):
34
+ raise ValueError(f"{meta_name} {key} changes across frames.")
35
+
36
+ width = int(camera_meta["w"])
37
+ height = int(camera_meta["h"])
38
+ cameras = []
39
+ seen_ids = set()
40
+ for camera_index, (k, w2c, camera_id) in enumerate(
41
+ zip(camera_meta["k"][0], camera_meta["w2c"][0], camera_meta["cam_id"][0])
42
+ ):
43
+ camera_id = int(camera_id)
44
+ if camera_id in seen_ids:
45
+ raise ValueError(f"{meta_name} lists camera {camera_id} more than once.")
46
+ seen_ids.add(camera_id)
47
+ filenames = [camera_meta["fn"][frame_index][camera_index] for frame_index in range(n_frames)]
48
+ w2c_tensor = torch.tensor(w2c, dtype=torch.float64)
49
+ quaternion = matrix_to_quaternion(w2c_tensor[:3, :3])
50
+ cameras.append(
51
+ ParsedCameraMeta(
52
+ camera=CameraModel(
53
+ name=f"cam{camera_id:02d}{Path(filenames[0]).suffix}",
54
+ width=width,
55
+ height=height,
56
+ fx=float(k[0][0]),
57
+ fy=float(k[1][1]),
58
+ cx=float(k[0][2]),
59
+ cy=float(k[1][2]),
60
+ qvec=quaternion.detach().cpu().numpy(),
61
+ tvec=w2c_tensor[:3, 3].detach().cpu().numpy(),
62
+ ),
63
+ filenames=filenames,
64
+ )
65
+ )
66
+ return cameras, n_frames
67
+
68
+
69
+ def read_camera_meta(
70
+ folder: Path,
71
+ include_train: bool = True,
72
+ include_test: bool = True,
73
+ ) -> tuple[list[ParsedCameraMeta], int]:
74
+ if not include_train and not include_test:
75
+ raise ValueError("At least one of train and test cameras must be included.")
76
+
77
+ n_frames = None
78
+ train_cameras, train_n_frames = [], None
79
+ if include_train:
80
+ train_cameras, train_n_frames = read_meta(folder, "train_meta.json")
81
+ n_frames = train_n_frames
82
+ test_cameras, test_n_frames = [], None
83
+ if include_test:
84
+ test_cameras, test_n_frames = read_meta(folder, "test_meta.json")
85
+ n_frames = test_n_frames
86
+
87
+ if include_train and include_test and train_n_frames != test_n_frames:
88
+ raise ValueError("train_meta.json and test_meta.json list different frame counts.")
89
+ cameras = train_cameras + test_cameras
90
+
91
+ merged: dict[str, ParsedCameraMeta] = {}
92
+ for camera in cameras:
93
+ if camera.camera.name in merged:
94
+ raise ValueError(f"Camera {camera.camera.name} appears in more than one metadata file.")
95
+ merged[camera.camera.name] = camera
96
+ if not merged:
97
+ raise ValueError("No cameras selected.")
98
+ return [merged[name] for name in sorted(merged)], n_frames
@@ -0,0 +1,52 @@
1
+ """Link Dynamic 3D Gaussians images and masks into per-frame folders."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ from pathlib import Path
7
+ from typing import Sequence
8
+
9
+ from .camera_meta import ParsedCameraMeta
10
+
11
+
12
+ def count_frame_dirs(output_pattern: Path, start_number: int = 1) -> int:
13
+ output_pattern = str(output_pattern)
14
+ frame = start_number
15
+ while Path(output_pattern % frame).is_dir():
16
+ frame += 1
17
+ n_frames = frame - start_number
18
+ if n_frames == 0:
19
+ raise FileNotFoundError(f"No frame directories found from pattern: {output_pattern}")
20
+ return n_frames
21
+
22
+
23
+ def link_file(src: Path, dst: Path) -> None:
24
+ if not src.is_file():
25
+ raise FileNotFoundError(f"Missing source file: {src}")
26
+ dst.parent.mkdir(parents=True, exist_ok=True)
27
+ if dst.exists():
28
+ dst.unlink()
29
+ os.link(src, dst)
30
+
31
+
32
+ def link_frames(
33
+ folder: Path,
34
+ frame_pattern: Path,
35
+ cameras: Sequence[ParsedCameraMeta],
36
+ n_frames: int,
37
+ start_number: int = 1,
38
+ image_dirname: str = "images",
39
+ ) -> None:
40
+ folder = Path(folder)
41
+ frame_pattern = str(frame_pattern)
42
+ for offset in range(n_frames):
43
+ frame_index = start_number - 1 + offset
44
+ frame_dir = Path(frame_pattern % (start_number + offset))
45
+ image_dir = frame_dir / image_dirname
46
+ mask_dir = frame_dir / "image_masks"
47
+ for camera in cameras:
48
+ filename = camera.filenames[frame_index]
49
+ link_file(folder / "ims" / filename, image_dir / camera.camera.name)
50
+ mask_src = folder / "seg" / Path(filename).with_suffix(".png")
51
+ if mask_src.is_file():
52
+ link_file(mask_src, mask_dir / f"{camera.camera.name}.png")
@@ -0,0 +1,117 @@
1
+ """Extract one or more videos into per-frame image folders."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ from pathlib import Path
7
+
8
+ from nvs2colmap.utils import extract_video_frames_parallel
9
+
10
+
11
+ def extract_videos(
12
+ video_paths: list[Path],
13
+ output_pattern: Path,
14
+ start_number: int = 1,
15
+ n_frames: int | None = None,
16
+ ffmpeg_executable: str = "ffmpeg",
17
+ ffprobe_executable: str = "ffprobe",
18
+ ffmpeg_processes: int = 1,
19
+ image_extension: str = ".png",
20
+ ) -> int:
21
+ if not image_extension.startswith("."):
22
+ image_extension = f".{image_extension}"
23
+
24
+ jobs = []
25
+ for index, video_path in enumerate(video_paths, start=1):
26
+ if not video_path.is_file():
27
+ raise FileNotFoundError(f"Missing video file: {video_path}")
28
+ jobs.append((video_path, str(output_pattern / f"{index:04d}{image_extension}")))
29
+
30
+ if not jobs:
31
+ raise ValueError("At least one video path is required.")
32
+
33
+ extracted_frame_counts = extract_video_frames_parallel(
34
+ jobs,
35
+ start_number=start_number,
36
+ n_frames=n_frames,
37
+ ffmpeg_executable=ffmpeg_executable,
38
+ ffprobe_executable=ffprobe_executable,
39
+ process_count=ffmpeg_processes,
40
+ )
41
+ return max(extracted_frame_counts)
42
+
43
+
44
+ def parse_args() -> argparse.Namespace:
45
+ parser = argparse.ArgumentParser(
46
+ description=(
47
+ "Extract videos into frame folders. For input videos v1.mp4 v2.mp4 "
48
+ "and --output-pattern 'frame%d/images', extracted images are named "
49
+ "frame1/images/0001.png, frame1/images/0002.png, etc."
50
+ )
51
+ )
52
+ parser.add_argument(
53
+ "videos",
54
+ type=Path,
55
+ nargs="+",
56
+ help="Video files to extract. Image names follow this input order.",
57
+ )
58
+ parser.add_argument(
59
+ "--output-pattern",
60
+ type=Path,
61
+ required=True,
62
+ help="Per-frame output directory pattern, for example 'frame%%d/images'.",
63
+ )
64
+ parser.add_argument(
65
+ "--n-frames",
66
+ type=int,
67
+ help="Number of frames to extract. Defaults to each video's available frame count.",
68
+ )
69
+ parser.add_argument(
70
+ "--start-number",
71
+ type=int,
72
+ default=1,
73
+ help="1-based source frame to start extracting from; output frame folders use the same starting number.",
74
+ )
75
+ parser.add_argument(
76
+ "--ffmpeg",
77
+ dest="ffmpeg_executable",
78
+ default="ffmpeg",
79
+ help="ffmpeg executable.",
80
+ )
81
+ parser.add_argument(
82
+ "--ffprobe",
83
+ dest="ffprobe_executable",
84
+ default="ffprobe",
85
+ help="ffprobe executable.",
86
+ )
87
+ parser.add_argument(
88
+ "--ffmpeg-processes",
89
+ type=int,
90
+ default=1,
91
+ help="Number of ffmpeg processes to run in parallel.",
92
+ )
93
+ parser.add_argument(
94
+ "--image-extension",
95
+ default=".png",
96
+ help="Image extension written by ffmpeg.",
97
+ )
98
+ return parser.parse_args()
99
+
100
+
101
+ def main() -> None:
102
+ args = parse_args()
103
+ n_frames = extract_videos(
104
+ video_paths=args.videos,
105
+ output_pattern=args.output_pattern,
106
+ start_number=args.start_number,
107
+ n_frames=args.n_frames,
108
+ ffmpeg_executable=args.ffmpeg_executable,
109
+ ffprobe_executable=args.ffprobe_executable,
110
+ ffmpeg_processes=args.ffmpeg_processes,
111
+ image_extension=args.image_extension,
112
+ )
113
+ print(f"Done: extracted up to {n_frames} frames per video.")
114
+
115
+
116
+ if __name__ == "__main__":
117
+ main()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: nvs2colmap
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Utilities for converting novel view synthesis datasets to COLMAP format.
5
5
  Author-email: Howard Yin <yindaheng98@gmail.com>
6
6
  Maintainer-email: Howard Yin <yindaheng98@gmail.com>
@@ -34,6 +34,8 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
34
34
 
35
35
  - **Neural 3D Video Dataset**: scenes with `poses_bounds.npy` and one `mp4`
36
36
  file per camera. See `nvs2colmap/n3dv/README.md`.
37
+ - **Dynamic 3D Gaussians**: scenes with `train_meta.json`, `test_meta.json`,
38
+ images in `ims/`, and masks in `seg/`. See `nvs2colmap/dynamic3dgs/README.md`.
37
39
 
38
40
  ## Supported Datasets
39
41
 
@@ -48,6 +50,10 @@ Utilities for converting novel view synthesis datasets to COLMAP format.
48
50
  [liuyubian/Robo360](https://huggingface.co/datasets/liuyubian/Robo360),
49
51
  paper
50
52
  [Robo360: A 3D Omnispective Multi-Material Robotic Manipulation Dataset](https://arxiv.org/abs/2312.06686).
53
+ - **Dynamic 3D Gaussians**: dataset
54
+ [JonathonLuiten/Dynamic3DGaussians](https://github.com/JonathonLuiten/Dynamic3DGaussians),
55
+ paper
56
+ [Dynamic 3D Gaussians: Tracking by Persistent Dynamic View Synthesis](https://arxiv.org/abs/2308.09713).
51
57
 
52
58
  ## Quick Start
53
59
 
@@ -88,18 +94,36 @@ Run the full COLMAP pipeline for each frame:
88
94
  ```bash
89
95
  python -m nvs2colmap.n3dv \
90
96
  --path data/Robo360/xarm6_gold_rope_in_basket_2 \
91
- --ffmpeg D:/MyPrograms/ffmpeg.exe \
92
- --ffprobe D:/MyPrograms/ffprobe.exe \
97
+ --ffmpeg ffmpeg \
98
+ --ffprobe ffprobe \
93
99
  --video-extension MP4 \
94
100
  --n-frames 1 \
95
101
  --use-colmap \
96
- --colmap-executable data/colmap/COLMAP.bat \
102
+ --colmap-executable colmap \
97
103
  --colmap-use-gpu 1
98
104
  ```
99
105
 
100
- By default, decoded frames are written to `frame*/images`, and the command also
101
- writes `frame*/sparse/0` text models. With `--use-colmap`, decoded frames are
102
- written to `frame*/input`, and each frame additionally gets the standard
103
- COLMAP outputs such as `distorted/`, `images/`, `sparse/`, and `stereo/`. When
104
- `--start-number N` is provided, decoding starts from source video frame `N`, and
105
- the generated folders/images are also numbered from `N`.
106
+ Link a Dynamic 3D Gaussians scene, including its masks, and write per-frame
107
+ COLMAP text models:
108
+
109
+ ```bash
110
+ python -m nvs2colmap.dynamic3dgs \
111
+ --path data/basketball \
112
+ --n-frames 150
113
+ ```
114
+
115
+ For Neural 3D Video scenes, decoded frames are written to `frame*/images` by
116
+ default, and the command also writes `frame*/sparse/0` text models. With
117
+ `--use-colmap`, decoded frames are written to `frame*/input`, and each frame
118
+ additionally gets the standard COLMAP outputs such as `distorted/`, `images/`,
119
+ `sparse/`, and `stereo/`. When `--start-number N` is provided, decoding starts
120
+ from source video frame `N`, and the generated folders/images are also numbered
121
+ from `N`.
122
+
123
+ For Dynamic 3D Gaussians scenes, images from both `train_meta.json` and
124
+ `test_meta.json` are hardlinked into `frame*/images`, and included masks are
125
+ hardlinked into `frame*/image_masks` (`cam01.jpg` pairs with `cam01.jpg.png`).
126
+ `--no-train-camera` and `--no-test-camera` drop one split. The same
127
+ `--use-colmap` switch writes images to `frame*/input` and runs COLMAP.
128
+ `--start-number` uses the same 1-based output numbering; source file
129
+ `000000.jpg` is frame `1`.
@@ -3,8 +3,11 @@ README.md
3
3
  pyproject.toml
4
4
  ./nvs2colmap/__init__.py
5
5
  ./nvs2colmap/colmap.py
6
+ ./nvs2colmap/extract_videos.py
6
7
  ./nvs2colmap/write_model.py
7
- ./nvs2colmap/dynamic3dgs/__init__.py
8
+ ./nvs2colmap/dynamic3dgs/__main__.py
9
+ ./nvs2colmap/dynamic3dgs/camera_meta.py
10
+ ./nvs2colmap/dynamic3dgs/link_frames.py
8
11
  ./nvs2colmap/n3dv/__main__.py
9
12
  ./nvs2colmap/n3dv/extract_videos.py
10
13
  ./nvs2colmap/n3dv/poses_bounds.py
@@ -15,6 +18,7 @@ pyproject.toml
15
18
  ./nvs2colmap/utils/rotation.py
16
19
  nvs2colmap/__init__.py
17
20
  nvs2colmap/colmap.py
21
+ nvs2colmap/extract_videos.py
18
22
  nvs2colmap/write_model.py
19
23
  nvs2colmap.egg-info/PKG-INFO
20
24
  nvs2colmap.egg-info/SOURCES.txt
@@ -22,7 +26,9 @@ nvs2colmap.egg-info/dependency_links.txt
22
26
  nvs2colmap.egg-info/entry_points.txt
23
27
  nvs2colmap.egg-info/requires.txt
24
28
  nvs2colmap.egg-info/top_level.txt
25
- nvs2colmap/dynamic3dgs/__init__.py
29
+ nvs2colmap/dynamic3dgs/__main__.py
30
+ nvs2colmap/dynamic3dgs/camera_meta.py
31
+ nvs2colmap/dynamic3dgs/link_frames.py
26
32
  nvs2colmap/n3dv/__main__.py
27
33
  nvs2colmap/n3dv/extract_videos.py
28
34
  nvs2colmap/n3dv/poses_bounds.py
@@ -0,0 +1,4 @@
1
+ [console_scripts]
2
+ nvs2colmap-dynamic3dgs = nvs2colmap.dynamic3dgs.__main__:main
3
+ nvs2colmap-extract-videos = nvs2colmap.extract_videos:main
4
+ nvs2colmap-n3dv = nvs2colmap.n3dv.__main__:main
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "nvs2colmap"
7
- version = "0.2.0"
7
+ version = "0.3.0"
8
8
  description = "Utilities for converting novel view synthesis datasets to COLMAP format."
9
9
  readme = "README.md"
10
10
  authors = [{ name = "Howard Yin", email = "yindaheng98@gmail.com" }]
@@ -41,7 +41,9 @@ Repository = "https://github.com/yindaheng98/NVS2COLMAP"
41
41
  Issues = "https://github.com/yindaheng98/NVS2COLMAP/issues"
42
42
 
43
43
  [project.scripts]
44
+ nvs2colmap-extract-videos = "nvs2colmap.extract_videos:main"
44
45
  nvs2colmap-n3dv = "nvs2colmap.n3dv.__main__:main"
46
+ nvs2colmap-dynamic3dgs = "nvs2colmap.dynamic3dgs.__main__:main"
45
47
 
46
48
  [tool.setuptools]
47
49
  packages = { find = { include = ["nvs2colmap*"] } }
File without changes
@@ -1,2 +0,0 @@
1
- [console_scripts]
2
- nvs2colmap-n3dv = nvs2colmap.n3dv.__main__:main
File without changes
File without changes