nvs2colmap 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
nvs2colmap/__init__.py ADDED
File without changes
nvs2colmap/colmap.py ADDED
@@ -0,0 +1,91 @@
1
+ """Run COLMAP processing for generated per-frame folders."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ from pathlib import Path
7
+ from typing import Sequence
8
+
9
+ from nvs2colmap.write_model import CameraModel
10
+ from nvs2colmap.utils.colmap import (
11
+ exhaustive_matcher,
12
+ feature_extractor,
13
+ image_undistorter,
14
+ mapper,
15
+ point_triangulator,
16
+ read_db,
17
+ )
18
+
19
+
20
+ def build_colmap_records(
21
+ cameras: Sequence[CameraModel],
22
+ image_extension: str = ".png",
23
+ ) -> tuple[dict[str, str], dict[str, str]]:
24
+ colmap_cameras, colmap_images = {}, {}
25
+ for camera in cameras:
26
+ img_name = f"{camera.name}{image_extension}"
27
+ colmap_cameras[img_name] = (
28
+ f"PINHOLE {camera.width} {camera.height} "
29
+ f"{camera.fx} {camera.fy} {camera.cx} {camera.cy}"
30
+ )
31
+ q, t = camera.qvec, camera.tvec
32
+ colmap_images[img_name] = f"{q[0]} {q[1]} {q[2]} {q[3]} {t[0]} {t[1]} {t[2]}"
33
+ return colmap_cameras, colmap_images
34
+
35
+
36
+ def run_colmap(
37
+ folder: Path,
38
+ cameras: Sequence[CameraModel],
39
+ image_extension: str = ".png",
40
+ colmap_executable: str = "colmap",
41
+ use_gpu: str = "1",
42
+ ) -> None:
43
+ folder = Path(folder)
44
+ colmap_executable = os.path.abspath(colmap_executable)
45
+ colmap_cameras, colmap_images = build_colmap_records(cameras, image_extension)
46
+
47
+ if feature_extractor(str(folder), use_gpu=use_gpu, colmap_executable=colmap_executable) != 0:
48
+ raise RuntimeError("Feature extraction failed")
49
+ if exhaustive_matcher(str(folder), use_gpu=use_gpu, colmap_executable=colmap_executable) != 0:
50
+ raise RuntimeError("Feature matching failed")
51
+
52
+ cam_ids, image_ids = read_db(str(folder))
53
+
54
+ mapper_input_path = folder / "distorted" / "sparse" / "loading"
55
+ os.makedirs(mapper_input_path, exist_ok=True)
56
+ with open(mapper_input_path / "cameras.txt", "w") as f:
57
+ for img_name, cam_id in sorted(cam_ids.items(), key=lambda i: i[1]):
58
+ f.write(f"{cam_id} {colmap_cameras[img_name]}\n")
59
+ with open(mapper_input_path / "images.txt", "w") as f:
60
+ for img_name, image_id in sorted(image_ids.items(), key=lambda i: i[1]):
61
+ f.write(f"{image_id} {colmap_images[img_name]} {cam_ids[img_name]} {img_name}\n\n")
62
+ open(mapper_input_path / "points3D.txt", "w").close()
63
+
64
+ if point_triangulator(str(folder), str(mapper_input_path), colmap_executable=colmap_executable) != 0:
65
+ raise RuntimeError("Triangulation failed")
66
+ if mapper(str(folder), str(mapper_input_path), colmap_executable=colmap_executable) != 0:
67
+ raise RuntimeError("Mapping failed")
68
+
69
+ # To fit sparse init in instantsplat
70
+ if image_undistorter(str(folder), colmap_executable=colmap_executable) != 0:
71
+ raise RuntimeError("Image undistortion failed")
72
+
73
+
74
+ def run_video_colmap(
75
+ output_pattern: Path,
76
+ cameras: Sequence[CameraModel],
77
+ n_frames: int,
78
+ start_number: int = 1,
79
+ image_extension: str = ".png",
80
+ colmap_executable: str = "colmap",
81
+ use_gpu: str = "1",
82
+ ) -> None:
83
+ output_pattern = str(output_pattern)
84
+ for frame in range(start_number, start_number + n_frames):
85
+ run_colmap(
86
+ folder=Path(output_pattern % frame),
87
+ cameras=cameras,
88
+ image_extension=image_extension,
89
+ colmap_executable=colmap_executable,
90
+ use_gpu=use_gpu,
91
+ )
File without changes
@@ -0,0 +1,126 @@
1
+ """Extract Neural 3D Video Dataset scenes into per-frame COLMAP folders."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ from pathlib import Path
7
+
8
+ from nvs2colmap.colmap import run_video_colmap
9
+ from nvs2colmap.write_model import write_video_colmap_text_model
10
+
11
+ from .extract_videos import count_frame_dirs, extract_videos
12
+ from .poses_bounds import read_poses_bounds
13
+
14
+
15
+ def parse_args() -> argparse.Namespace:
16
+ parser = argparse.ArgumentParser(
17
+ description=(
18
+ "Extract a Neural 3D Video Dataset scene and convert poses_bounds.npy "
19
+ "to per-frame COLMAP text models."
20
+ )
21
+ )
22
+ parser.add_argument(
23
+ "--path",
24
+ type=Path,
25
+ required=True,
26
+ help="Scene directory containing poses_bounds.npy and camera mp4 files.",
27
+ )
28
+ parser.add_argument(
29
+ "--n-frames",
30
+ type=int,
31
+ help="Number of frames to extract and write. Defaults to the video frame count when extracting.",
32
+ )
33
+ parser.add_argument(
34
+ "--ffmpeg",
35
+ dest="ffmpeg_executable",
36
+ default="ffmpeg",
37
+ help="ffmpeg executable.",
38
+ )
39
+ parser.add_argument(
40
+ "--ffprobe",
41
+ dest="ffprobe_executable",
42
+ default="ffprobe",
43
+ help="ffprobe executable.",
44
+ )
45
+ parser.add_argument(
46
+ "--image-extension",
47
+ default=".png",
48
+ help="Image extension written by ffmpeg and referenced by COLMAP text files.",
49
+ )
50
+ parser.add_argument(
51
+ "--video-extension",
52
+ default=".mp4",
53
+ help="Video extension used to discover camera videos.",
54
+ )
55
+ parser.add_argument(
56
+ "--skip-video-extraction",
57
+ action="store_true",
58
+ help="Only convert poses_bounds.npy for existing frame folders.",
59
+ )
60
+ parser.add_argument(
61
+ "--use-colmap",
62
+ action="store_true",
63
+ help=(
64
+ "Run COLMAP feature extraction, matching, triangulation, mapping, "
65
+ "and undistortion instead of only writing sparse/0 text models."
66
+ ),
67
+ )
68
+ parser.add_argument(
69
+ "--colmap-executable",
70
+ default="colmap",
71
+ help="COLMAP executable used when --use-colmap is set.",
72
+ )
73
+ parser.add_argument(
74
+ "--colmap-use-gpu",
75
+ dest="colmap_use_gpu",
76
+ default="1",
77
+ help="Whether COLMAP SIFT extraction/matching should use GPU when --use-colmap is set.",
78
+ )
79
+ return parser.parse_args()
80
+
81
+
82
+ def main() -> None:
83
+ args = parse_args()
84
+ folder = args.path.resolve()
85
+
86
+ cameras = read_poses_bounds(folder, args.video_extension)
87
+
88
+ n_frames = args.n_frames
89
+ frame_output_pattern = folder / "frame%d"
90
+ image_dir_name = "input" if args.use_colmap else "images"
91
+ if not args.skip_video_extraction:
92
+ n_frames = extract_videos(
93
+ folder=folder,
94
+ output_pattern=frame_output_pattern / image_dir_name,
95
+ cameras=cameras,
96
+ n_frames=n_frames,
97
+ ffmpeg_executable=args.ffmpeg_executable,
98
+ ffprobe_executable=args.ffprobe_executable,
99
+ video_extension=args.video_extension,
100
+ image_extension=args.image_extension,
101
+ )
102
+ elif n_frames is None:
103
+ n_frames = count_frame_dirs(frame_output_pattern)
104
+
105
+ if not args.use_colmap:
106
+ write_video_colmap_text_model(
107
+ output_pattern=frame_output_pattern / "sparse" / "0",
108
+ cameras=cameras,
109
+ n_frames=n_frames,
110
+ image_extension=args.image_extension,
111
+ )
112
+ else:
113
+ run_video_colmap(
114
+ output_pattern=frame_output_pattern,
115
+ cameras=cameras,
116
+ n_frames=n_frames,
117
+ image_extension=args.image_extension,
118
+ colmap_executable=args.colmap_executable,
119
+ use_gpu=args.colmap_use_gpu,
120
+ )
121
+
122
+ print(f"Done: {folder}")
123
+
124
+
125
+ if __name__ == "__main__":
126
+ main()
@@ -0,0 +1,49 @@
1
+ """Extract Neural 3D Video camera videos into frame image folders."""
2
+
3
+ from pathlib import Path
4
+
5
+ from nvs2colmap.write_model import CameraModel
6
+ from nvs2colmap.utils import extract_video_frames
7
+
8
+
9
+ def count_frame_dirs(output_pattern: Path, start_number: int = 1) -> int:
10
+ output_pattern = str(output_pattern)
11
+ frame = start_number
12
+ while Path(output_pattern % frame).is_dir():
13
+ frame += 1
14
+ n_frames = frame - start_number
15
+ if n_frames == 0:
16
+ raise FileNotFoundError(f"No frame directories found from pattern: {output_pattern}")
17
+ return n_frames
18
+
19
+
20
+ def extract_videos(
21
+ folder: Path,
22
+ output_pattern: Path,
23
+ cameras: list[CameraModel],
24
+ n_frames: int | None = None,
25
+ ffmpeg_executable: str = "ffmpeg",
26
+ ffprobe_executable: str = "ffprobe",
27
+ video_extension: str = ".mp4",
28
+ image_extension: str = ".png",
29
+ ) -> int:
30
+ if not video_extension.startswith("."):
31
+ video_extension = f".{video_extension}"
32
+
33
+ extracted_n_frames = 0
34
+ for camera in cameras:
35
+ video_path = folder / f"{camera.name}{video_extension}"
36
+ if not video_path.is_file():
37
+ raise FileNotFoundError(f"Missing camera video: {video_path}")
38
+ camera_output_pattern = output_pattern / f"{camera.name}{image_extension}"
39
+ camera_n_frames = extract_video_frames(
40
+ video_path=video_path,
41
+ output_pattern=str(camera_output_pattern),
42
+ n_frames=n_frames,
43
+ ffmpeg_executable=ffmpeg_executable,
44
+ ffprobe_executable=ffprobe_executable,
45
+ )
46
+ extracted_n_frames = max(extracted_n_frames, camera_n_frames)
47
+ if extracted_n_frames == 0:
48
+ raise ValueError("No cameras to extract.")
49
+ return extracted_n_frames
@@ -0,0 +1,73 @@
1
+ """Read Neural 3D Video poses_bounds.npy camera metadata."""
2
+
3
+ import numpy as np
4
+ import torch
5
+ from pathlib import Path
6
+
7
+ from nvs2colmap.write_model import CameraModel
8
+ from nvs2colmap.utils import matrix_to_quaternion
9
+
10
+
11
+ def read_camera_meta_n3dv(folder):
12
+ folder = Path(folder)
13
+ # Inverse of: https://github.com/Fyusion/LLFF/blob/master/llff/poses/pose_utils.py#L11
14
+ poses_arr = torch.tensor(np.load(folder / "poses_bounds.npy"))
15
+ poses = poses_arr[:, :-2].reshape(-1, 3, 5)
16
+ bds = poses_arr[:, -2:]
17
+ hwf = poses[:, :, 4]
18
+ c2w = torch.zeros((poses.shape[0], 4, 4), dtype=poses.dtype)
19
+ # switch from [-y, x, z] (poses_bounds format) back to [x, -y, -z] (colmap format)
20
+ c2w[:, :3, 0:1] = poses[:, :3, 1:2]
21
+ c2w[:, :3, 1:2] = poses[:, :3, 0:1]
22
+ c2w[:, :3, 2:3] = -poses[:, :3, 2:3]
23
+ c2w[:, :3, 3:4] = poses[:, :3, 3:4]
24
+ c2w[:, 3, 3] = 1
25
+ w2c = torch.linalg.inv(c2w)
26
+ Rs = w2c[:, :3, :3]
27
+ Ts = w2c[:, :3, 3]
28
+ return poses.shape[0], Rs, Ts, hwf, bds
29
+
30
+
31
+ def list_camera_videos(folder: Path, video_extension: str = ".mp4") -> list[Path]:
32
+ if not video_extension.startswith("."):
33
+ video_extension = f".{video_extension}"
34
+ video_extension = video_extension.lower()
35
+ videos = [
36
+ path
37
+ for path in folder.iterdir()
38
+ if path.is_file() and path.suffix.lower() == video_extension
39
+ ]
40
+ if not videos:
41
+ raise FileNotFoundError(f"No {video_extension} camera videos found in {folder}")
42
+ return sorted(videos)
43
+
44
+
45
+ def read_poses_bounds(folder, video_extension: str = ".mp4") -> list[CameraModel]:
46
+ folder = Path(folder)
47
+ camera_meta = read_camera_meta_n3dv(folder)
48
+ n_cameras, Rs, Ts, hwf, bds = camera_meta
49
+ camera_videos = list_camera_videos(folder, video_extension)
50
+ if len(camera_videos) != n_cameras:
51
+ raise ValueError(f"Expected {n_cameras} camera videos, got {len(camera_videos)}.")
52
+
53
+ cameras = []
54
+ for i in range(n_cameras):
55
+ height, width = hwf[i, 0], hwf[i, 1]
56
+ fx = fy = hwf[i, 2]
57
+ cx, cy = width / 2, height / 2
58
+ R, T = Rs[i], Ts[i]
59
+ q, t = matrix_to_quaternion(R), T
60
+ cameras.append(
61
+ CameraModel(
62
+ name=camera_videos[i].stem,
63
+ width=int(round(float(width))),
64
+ height=int(round(float(height))),
65
+ fx=float(fx),
66
+ fy=float(fy),
67
+ cx=float(cx),
68
+ cy=float(cy),
69
+ qvec=q.detach().cpu().numpy(),
70
+ tvec=t.detach().cpu().numpy(),
71
+ )
72
+ )
73
+ return cameras
File without changes
@@ -0,0 +1,31 @@
1
+ """Utility helpers exported for package-level imports."""
2
+
3
+ from .colmap import (
4
+ execute,
5
+ exhaustive_matcher,
6
+ feature_extractor,
7
+ image_undistorter,
8
+ mapper,
9
+ model_converter_bin,
10
+ model_converter_txt,
11
+ point_triangulator,
12
+ read_db,
13
+ )
14
+ from .ffmpeg import count_video_frames, extract_video_frames
15
+ from .rotation import matrix_to_quaternion, standardize_quaternion
16
+
17
+ __all__ = [
18
+ "count_video_frames",
19
+ "execute",
20
+ "exhaustive_matcher",
21
+ "extract_video_frames",
22
+ "feature_extractor",
23
+ "image_undistorter",
24
+ "mapper",
25
+ "matrix_to_quaternion",
26
+ "model_converter_bin",
27
+ "model_converter_txt",
28
+ "point_triangulator",
29
+ "read_db",
30
+ "standardize_quaternion",
31
+ ]
@@ -0,0 +1,104 @@
1
+ """COLMAP command helpers."""
2
+
3
+ import sqlite3
4
+ import subprocess
5
+ import os
6
+
7
+
8
+ def execute(cmd):
9
+ proc = subprocess.Popen(cmd, shell=False)
10
+ proc.communicate()
11
+ return proc.returncode
12
+
13
+
14
+ def feature_extractor(folder, use_gpu="1", colmap_executable="colmap"):
15
+ os.makedirs(os.path.join(folder, "distorted"), exist_ok=True)
16
+ cmd = [
17
+ colmap_executable, "feature_extractor",
18
+ "--database_path", os.path.join(folder, "distorted", "database.db"),
19
+ "--image_path", os.path.join(folder, "input"),
20
+ "--ImageReader.camera_model", "PINHOLE",
21
+ "--SiftExtraction.use_gpu", use_gpu,
22
+ "--ImageReader.single_camera_per_image", "1",
23
+ ]
24
+ return execute(cmd)
25
+
26
+
27
+ def exhaustive_matcher(folder, use_gpu="1", colmap_executable="colmap"):
28
+ cmd = [
29
+ colmap_executable, "exhaustive_matcher",
30
+ "--database_path", os.path.join(folder, "distorted", "database.db"),
31
+ "--SiftMatching.use_gpu", use_gpu,
32
+ ]
33
+ return execute(cmd)
34
+
35
+
36
+ def read_db(folder):
37
+ conn = sqlite3.connect(os.path.join(folder, "distorted", "database.db"))
38
+ c = conn.cursor()
39
+ c.execute(f"SELECT camera_id,image_id,name FROM main.images")
40
+ camera_ids, image_ids = {}, {}
41
+ for camera_id, image_id, name in c.fetchall():
42
+ camera_ids[name] = camera_id
43
+ image_ids[name] = image_id
44
+ conn.close()
45
+ return camera_ids, image_ids
46
+
47
+
48
+ def point_triangulator(folder, mapper_input_path, colmap_executable="colmap"):
49
+ cmd = [
50
+ colmap_executable, "point_triangulator",
51
+ "--database_path", os.path.join(folder, "distorted", "database.db"),
52
+ "--input_path", mapper_input_path,
53
+ "--output_path", mapper_input_path,
54
+ "--image_path", os.path.join(folder, "input")
55
+ ]
56
+ return execute(cmd)
57
+
58
+
59
+ def mapper(folder, mapper_input_path, colmap_executable="colmap"):
60
+ os.makedirs(os.path.join(folder, "distorted", "sparse", "0"), exist_ok=True)
61
+ cmd = [
62
+ colmap_executable, "mapper",
63
+ "--database_path", os.path.join(folder, "distorted", "database.db"),
64
+ "--image_path", os.path.join(folder, "input"),
65
+ "--Mapper.ba_global_function_tolerance=0.000001",
66
+ "--input_path", mapper_input_path,
67
+ "--output_path", os.path.join(folder, "distorted", "sparse", "0")
68
+ ]
69
+ return execute(cmd)
70
+
71
+
72
+ def model_converter_txt(folder, colmap_executable="colmap"):
73
+ mapper_output_path = os.path.join(folder, "distorted", "sparse", "0")
74
+ os.makedirs(mapper_output_path, exist_ok=True)
75
+ cmd = [
76
+ colmap_executable, "model_converter",
77
+ "--input_path", mapper_output_path,
78
+ "--output_path", mapper_output_path,
79
+ "--output_type=TXT",
80
+ ]
81
+ return execute(cmd)
82
+
83
+
84
+ def model_converter_bin(folder, colmap_executable="colmap"):
85
+ mapper_output_path = os.path.join(folder, "distorted", "sparse", "0")
86
+ os.makedirs(mapper_output_path, exist_ok=True)
87
+ cmd = [
88
+ colmap_executable, "model_converter",
89
+ "--input_path", mapper_output_path,
90
+ "--output_path", mapper_output_path,
91
+ "--output_type=BIN",
92
+ ]
93
+ return execute(cmd)
94
+
95
+
96
+ def image_undistorter(folder, colmap_executable="colmap"):
97
+ cmd = [
98
+ colmap_executable, "image_undistorter",
99
+ "--image_path", os.path.join(folder, "input"),
100
+ "--input_path", os.path.join(folder, "distorted", "sparse", "0"),
101
+ "--output_path", folder,
102
+ "--output_type=COLMAP",
103
+ ]
104
+ return execute(cmd)
@@ -0,0 +1,68 @@
1
+ """ffmpeg helpers."""
2
+
3
+ import os
4
+ import subprocess
5
+
6
+
7
+ def count_video_frames(
8
+ video_path,
9
+ start_number: int = 1,
10
+ n_frames: int | None = None,
11
+ ffprobe_executable: str = "ffprobe",
12
+ ) -> int:
13
+ assert start_number >= 1, f"start_number must be >= 1, got {start_number}."
14
+
15
+ cmd = [
16
+ ffprobe_executable,
17
+ "-v",
18
+ "error",
19
+ "-count_frames",
20
+ "-select_streams",
21
+ "v:0",
22
+ "-show_entries",
23
+ "stream=nb_read_frames",
24
+ "-of",
25
+ "default=nokey=1:noprint_wrappers=1",
26
+ str(video_path),
27
+ ]
28
+ output = subprocess.check_output(cmd, text=True).strip()
29
+ video_frame_count = int(output)
30
+ available_frame_count = video_frame_count - start_number + 1
31
+ assert available_frame_count > 0, f"start_number {start_number} is outside {video_frame_count} frames in {video_path}."
32
+
33
+ if n_frames is None:
34
+ return available_frame_count
35
+
36
+ assert n_frames >= 1, f"n_frames must be >= 1, got {n_frames}."
37
+ return min(n_frames, available_frame_count)
38
+
39
+
40
+ def extract_video_frames(
41
+ video_path, output_pattern,
42
+ start_number: int = 1, n_frames: int | None = None,
43
+ ffmpeg_executable: str = "ffmpeg", ffprobe_executable: str = "ffprobe",
44
+ ) -> int:
45
+ frame_count = count_video_frames(
46
+ video_path,
47
+ start_number=start_number,
48
+ n_frames=n_frames,
49
+ ffprobe_executable=ffprobe_executable,
50
+ )
51
+
52
+ for frame in range(start_number, start_number + frame_count):
53
+ os.makedirs(os.path.dirname(output_pattern % frame), exist_ok=True)
54
+
55
+ cmd = [
56
+ ffmpeg_executable,
57
+ "-hide_banner",
58
+ "-loglevel",
59
+ "error",
60
+ "-i",
61
+ str(video_path),
62
+ "-frames:v", str(frame_count),
63
+ "-start_number", str(start_number),
64
+ str(output_pattern), "-y",
65
+ ]
66
+ print(" ".join(str(part) for part in cmd))
67
+ subprocess.run(cmd, check=True)
68
+ return frame_count
@@ -0,0 +1,99 @@
1
+ """Rotation conversion utilities."""
2
+
3
+ import torch
4
+ import torch.nn.functional as F
5
+
6
+
7
+ def standardize_quaternion(quaternions: torch.Tensor) -> torch.Tensor:
8
+ """
9
+ Convert a unit quaternion to a standard form: one in which the real
10
+ part is non negative.
11
+
12
+ Args:
13
+ quaternions: Quaternions with real part first,
14
+ as tensor of shape (..., 4).
15
+
16
+ Returns:
17
+ Standardized quaternions as tensor of shape (..., 4).
18
+ Source: https://pytorch3d.readthedocs.io/en/latest/_modules/pytorch3d/transforms/rotation_conversions.html
19
+ """
20
+ return torch.where(quaternions[..., 0:1] < 0, -quaternions, quaternions)
21
+
22
+
23
+ def _sqrt_positive_part(x: torch.Tensor) -> torch.Tensor:
24
+ """
25
+ Returns torch.sqrt(torch.max(0, x))
26
+ but with a zero subgradient where x is 0.
27
+ Source: https://pytorch3d.readthedocs.io/en/latest/_modules/pytorch3d/transforms/rotation_conversions.html
28
+ """
29
+ ret = torch.zeros_like(x)
30
+ positive_mask = x > 0
31
+ # if torch.is_grad_enabled():
32
+ # ret[positive_mask] = torch.sqrt(x[positive_mask])
33
+ # else:
34
+ # ret = torch.where(positive_mask, torch.sqrt(x), ret)
35
+ ret[positive_mask] = torch.sqrt(x[positive_mask])
36
+ return ret
37
+
38
+
39
+ def matrix_to_quaternion(matrix: torch.Tensor) -> torch.Tensor:
40
+ """
41
+ Convert rotations given as rotation matrices to quaternions.
42
+
43
+ Args:
44
+ matrix: Rotation matrices as tensor of shape (..., 3, 3).
45
+
46
+ Returns:
47
+ quaternions with real part first, as tensor of shape (..., 4).
48
+ Source: https://pytorch3d.readthedocs.io/en/latest/_modules/pytorch3d/transforms/rotation_conversions.html
49
+ """
50
+ if matrix.size(-1) != 3 or matrix.size(-2) != 3:
51
+ raise ValueError(f"Invalid rotation matrix shape {matrix.shape}.")
52
+
53
+ batch_dim = matrix.shape[:-2]
54
+ m00, m01, m02, m10, m11, m12, m20, m21, m22 = torch.unbind(
55
+ matrix.reshape(batch_dim + (9,)), dim=-1
56
+ )
57
+
58
+ q_abs = _sqrt_positive_part(
59
+ torch.stack(
60
+ [
61
+ 1.0 + m00 + m11 + m22,
62
+ 1.0 + m00 - m11 - m22,
63
+ 1.0 - m00 + m11 - m22,
64
+ 1.0 - m00 - m11 + m22,
65
+ ],
66
+ dim=-1,
67
+ )
68
+ )
69
+
70
+ # we produce the desired quaternion multiplied by each of r, i, j, k
71
+ quat_by_rijk = torch.stack(
72
+ [
73
+ # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and
74
+ # `int`.
75
+ torch.stack([q_abs[..., 0] ** 2, m21 - m12, m02 - m20, m10 - m01], dim=-1),
76
+ # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and
77
+ # `int`.
78
+ torch.stack([m21 - m12, q_abs[..., 1] ** 2, m10 + m01, m02 + m20], dim=-1),
79
+ # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and
80
+ # `int`.
81
+ torch.stack([m02 - m20, m10 + m01, q_abs[..., 2] ** 2, m12 + m21], dim=-1),
82
+ # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and
83
+ # `int`.
84
+ torch.stack([m10 - m01, m20 + m02, m21 + m12, q_abs[..., 3] ** 2], dim=-1),
85
+ ],
86
+ dim=-2,
87
+ )
88
+
89
+ # We floor here at 0.1 but the exact level is not important; if q_abs is small,
90
+ # the candidate won't be picked.
91
+ flr = torch.tensor(0.1).to(dtype=q_abs.dtype, device=q_abs.device)
92
+ quat_candidates = quat_by_rijk / (2.0 * q_abs[..., None].max(flr))
93
+
94
+ # if not for numerical problems, quat_candidates[i] should be same (up to a sign),
95
+ # forall i; we pick the best-conditioned one (with the largest denominator)
96
+ out = quat_candidates[
97
+ F.one_hot(q_abs.argmax(dim=-1), num_classes=4) > 0.5, :
98
+ ].reshape(batch_dim + (4,))
99
+ return standardize_quaternion(out)
@@ -0,0 +1,67 @@
1
+ """Base COLMAP data structures and text model writers."""
2
+
3
+ import os
4
+ from dataclasses import dataclass
5
+ from typing import Sequence
6
+
7
+ import numpy as np
8
+
9
+
10
+ @dataclass(frozen=True)
11
+ class CameraModel:
12
+ """A static camera model converted from one row of poses_bounds.npy."""
13
+
14
+ name: str
15
+ width: int
16
+ height: int
17
+ fx: float
18
+ fy: float
19
+ cx: float
20
+ cy: float
21
+ qvec: np.ndarray
22
+ tvec: np.ndarray
23
+
24
+
25
+ def write_colmap_text_model(
26
+ path,
27
+ cameras: Sequence[CameraModel],
28
+ image_extension: str = ".png",
29
+ ) -> None:
30
+ colmap_cameras, colmap_images = {}, {}
31
+ for camera in cameras:
32
+ img_name = f"{camera.name}{image_extension}"
33
+ height, width = camera.height, camera.width
34
+ fx, fy = camera.fx, camera.fy
35
+ cx, cy = camera.cx, camera.cy
36
+ colmap_cameras[img_name] = f"PINHOLE {width} {height} {fx} {fy} {cx} {cy}"
37
+ q, t = camera.qvec, camera.tvec
38
+ colmap_images[img_name] = f"{q[0]} {q[1]} {q[2]} {q[3]} {t[0]} {t[1]} {t[2]}"
39
+
40
+ cam_ids = {f"{camera.name}{image_extension}": i for i, camera in enumerate(cameras, start=1)}
41
+ image_ids = {f"{camera.name}{image_extension}": i for i, camera in enumerate(cameras, start=1)}
42
+
43
+ mapper_input_path = path
44
+ os.makedirs(mapper_input_path, exist_ok=True)
45
+ with open(os.path.join(mapper_input_path, "cameras.txt"), "w") as f:
46
+ for img_name, cam_id in sorted(cam_ids.items(), key=lambda i: i[1]):
47
+ f.write(f"{cam_id} {colmap_cameras[img_name]}\n")
48
+ with open(os.path.join(mapper_input_path, "images.txt"), "w") as f:
49
+ for img_name, image_id in sorted(image_ids.items(), key=lambda i: i[1]):
50
+ f.write(f"{image_id} {colmap_images[img_name]} {cam_ids[img_name]} {img_name}\n\n")
51
+ open(os.path.join(mapper_input_path, "points3D.txt"), "w").close()
52
+
53
+
54
+ def write_video_colmap_text_model(
55
+ output_pattern,
56
+ cameras: Sequence[CameraModel],
57
+ n_frames: int,
58
+ start_number: int = 1,
59
+ image_extension: str = ".png",
60
+ ) -> None:
61
+ output_pattern = str(output_pattern)
62
+ for frame in range(start_number, start_number + n_frames):
63
+ write_colmap_text_model(
64
+ output_pattern % frame,
65
+ cameras,
66
+ image_extension=image_extension,
67
+ )
@@ -0,0 +1,92 @@
1
+ Metadata-Version: 2.4
2
+ Name: nvs2colmap
3
+ Version: 0.1.0
4
+ Summary: Utilities for converting novel view synthesis datasets to COLMAP format.
5
+ Author-email: Howard Yin <yindaheng98@gmail.com>
6
+ Maintainer-email: Howard Yin <yindaheng98@gmail.com>
7
+ Project-URL: Homepage, https://github.com/yindaheng98/NVS2COLMAP
8
+ Project-URL: Documentation, https://github.com/yindaheng98/NVS2COLMAP#readme
9
+ Project-URL: Repository, https://github.com/yindaheng98/NVS2COLMAP
10
+ Project-URL: Issues, https://github.com/yindaheng98/NVS2COLMAP/issues
11
+ Keywords: colmap,novel-view-synthesis,neural-3d-video,3d-reconstruction,computer-vision
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
+ Classifier: Topic :: Scientific/Engineering :: Image Processing
22
+ Requires-Python: >=3.10
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ Requires-Dist: numpy
26
+ Requires-Dist: torch
27
+ Dynamic: license-file
28
+
29
+ # NVS2COLMAP
30
+
31
+ Utilities for converting novel view synthesis datasets to COLMAP format.
32
+
33
+ ## Supported Formats
34
+
35
+ - **Neural 3D Video Dataset**: scenes with `poses_bounds.npy` and one `mp4`
36
+ file per camera. See `nvs2colmap/n3dv/README.md`.
37
+
38
+ ## Supported Datasets
39
+
40
+ - **Neural 3D Video Dataset**: dataset
41
+ [facebookresearch/Neural_3D_Video](https://github.com/facebookresearch/Neural_3D_Video),
42
+ paper
43
+ [Neural 3D Video Synthesis from Multi-view Video](https://arxiv.org/abs/2103.02597).
44
+ - **StreamRF / Meet Room Dataset**: dataset
45
+ [AlgoHunt/StreamRF](https://github.com/AlgoHunt/StreamRF), paper
46
+ [Streaming Radiance Fields for 3D Video Synthesis](https://arxiv.org/abs/2210.14831).
47
+ - **Robo360**: dataset
48
+ [liuyubian/Robo360](https://huggingface.co/datasets/liuyubian/Robo360),
49
+ paper
50
+ [Robo360: A 3D Omnispective Multi-Material Robotic Manipulation Dataset](https://arxiv.org/abs/2312.06686).
51
+
52
+ ## Quick Start
53
+
54
+ Install the Python runtime dependencies:
55
+
56
+ ```bash
57
+ pip install numpy torch
58
+ ```
59
+
60
+ For Neural 3D Video scenes, the command also needs `ffmpeg` and `ffprobe` on
61
+ `PATH`, or explicit paths via `--ffmpeg` and `--ffprobe`. If you want to run
62
+ the full COLMAP pipeline, also provide a COLMAP executable via
63
+ `--colmap-executable`.
64
+
65
+ Extract a Neural 3D Video scene and write per-frame COLMAP text models:
66
+
67
+ ```bash
68
+ python -m nvs2colmap.n3dv \
69
+ --path data/coffee_martini \
70
+ --ffmpeg ffmpeg \
71
+ --ffprobe ffprobe \
72
+ --n-frames 300
73
+ ```
74
+
75
+ Run the full COLMAP pipeline for each frame:
76
+
77
+ ```bash
78
+ python -m nvs2colmap.n3dv \
79
+ --path data/Robo360/xarm6_gold_rope_in_basket_2 \
80
+ --ffmpeg D:/MyPrograms/ffmpeg.exe \
81
+ --ffprobe D:/MyPrograms/ffprobe.exe \
82
+ --video-extension MP4 \
83
+ --n-frames 1 \
84
+ --use-colmap \
85
+ --colmap-executable data/colmap/COLMAP.bat \
86
+ --colmap-use-gpu 1
87
+ ```
88
+
89
+ By default, decoded frames are written to `frame*/images`, and the command also
90
+ writes `frame*/sparse/0` text models. With `--use-colmap`, decoded frames are
91
+ written to `frame*/input`, and each frame additionally gets the standard
92
+ COLMAP outputs such as `distorted/`, `images/`, `sparse/`, and `stereo/`.
@@ -0,0 +1,18 @@
1
+ nvs2colmap/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
2
+ nvs2colmap/colmap.py,sha256=gWXjLhCq6bXd9djuJMTDe5ityXl8DQvNM6rnbGKOR0o,3334
3
+ nvs2colmap/write_model.py,sha256=ku4l1xBOU8yibrZJvYX32pluLwBovlEepMcPsCj35hk,2244
4
+ nvs2colmap/dynamic3dgs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
5
+ nvs2colmap/n3dv/__main__.py,sha256=4vBIay787OqKhL3P_DGKr45GwZ1T5ay8qMyolsNhrb4,3851
6
+ nvs2colmap/n3dv/extract_videos.py,sha256=9cngqMXddzVswiYD2gMWvE5KzjMrPrJlcqCSqOxFDvQ,1735
7
+ nvs2colmap/n3dv/poses_bounds.py,sha256=_m0i-vrwW9M5yWFFExKFwjLMtyC13myUVcCg_rAjBh4,2572
8
+ nvs2colmap/stnerf/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
9
+ nvs2colmap/utils/__init__.py,sha256=xzEaKJoUQgx8V6swE2XA0XgE7lCpIbhOQ1tQu_ePHHM,720
10
+ nvs2colmap/utils/colmap.py,sha256=ubfICJLQLc2ZpnntR4sdVMDLo8Xe4nOSGadPwAVfIrA,3441
11
+ nvs2colmap/utils/ffmpeg.py,sha256=hJD9OXx7Qe_gdqGnE0jcuqVoRG8NOSFMTIe2VOlaMKQ,1928
12
+ nvs2colmap/utils/rotation.py,sha256=0kX_jUXwP1Zi305AaKDyLdkIY9rWaldE70Wz7zEE_MM,3701
13
+ nvs2colmap-0.1.0.dist-info/licenses/LICENSE,sha256=FVon4L020BfAcR4lOrma9lvbMf-6AYGcsXD9uKKFAlk,1066
14
+ nvs2colmap-0.1.0.dist-info/METADATA,sha256=xKA7BmmdteT2ObUZ4JuVf1xknCA9XwAPGQm5dtTxbVY,3369
15
+ nvs2colmap-0.1.0.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
16
+ nvs2colmap-0.1.0.dist-info/entry_points.txt,sha256=PbwBOs0j_u2YXvZ1kYeoVExCoeTU5BL9VpJhZZJBZL4,66
17
+ nvs2colmap-0.1.0.dist-info/top_level.txt,sha256=SpoGd8Uz0ksy9BkVHl2e6QLQtW6s5YarGkAoTTjDp2k,11
18
+ nvs2colmap-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (82.0.1)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ nvs2colmap-n3dv = nvs2colmap.n3dv.__main__:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2022 Howard Yin
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ nvs2colmap