Repository navigation
Expand file tree
/
Copy pathconvert.py
More file actions
280 lines (255 loc) · 13.5 KB
/
Copy pathconvert.py
File metadata and controls
280 lines (255 loc) · 13.5 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
#
# Copyright (C) 2023, Inria
# GRAPHDECO research group, https://team.inria.fr/graphdeco
# All rights reserved.
#
# This software is free for non-commercial, research and evaluation use
# under the terms of the LICENSE.md file.
#
# For inquiries contact george.drettakis@inria.fr
#
import os
import logging
from argparse import ArgumentParser
import shutil
# This Python script is based on the shell converter script provided in the MipNerF 360 repository.
parser = ArgumentParser("Colmap converter")
parser.add_argument("--no_gpu", action='store_true')
parser.add_argument("--skip_matching", action='store_true')
parser.add_argument("--source_path", "-s", required=True, type=str)
parser.add_argument("--camera", default="OPENCV", type=str)
parser.add_argument("--colmap_executable", default="", type=str)
parser.add_argument("--resize", action="store_true")
parser.add_argument("--magick_executable", default="", type=str)
parser.add_argument("--long_edge", default=0, type=int,
help="After undistortion, rewrite images/ and sparse/0/cameras.bin "
"at this long edge, IN PLACE. 512 is what every number in the "
"paper was measured at. 0 = leave the scene at full size.")
parser.add_argument("--only_resize", action="store_true",
help="Skip COLMAP entirely and only apply --long_edge to a scene that "
"already has images/ and sparse/0/ with PINHOLE intrinsics.")
parser.add_argument("--from_nerfstudio", default="", type=str,
help="Ingest a scene shipped in the nerfstudio layout "
"(<dir>/colmap/sparse/0 plus <dir>/images_N), undistorting "
"OPENCV -> PINHOLE, into -s as a plain COLMAP scene. This is "
"the format the DL3DV release uses.")
parser.add_argument("--images_dir", default="images_4", type=str,
help="Which image directory inside --from_nerfstudio to read.")
args = parser.parse_args()
if args.only_resize or args.from_nerfstudio:
args.skip_matching = True
if args.only_resize and not args.long_edge:
parser.error("--only_resize needs --long_edge, e.g. --only_resize --long_edge 512")
colmap_command = '"{}"'.format(args.colmap_executable) if len(args.colmap_executable) > 0 else "colmap"
magick_command = '"{}"'.format(args.magick_executable) if len(args.magick_executable) > 0 else "magick"
use_gpu = 1 if not args.no_gpu else 0
if not args.skip_matching:
os.makedirs(args.source_path + "/distorted/sparse", exist_ok=True)
## Feature extraction
feat_extracton_cmd = colmap_command + " feature_extractor "\
"--database_path " + args.source_path + "/distorted/database.db \
--image_path " + args.source_path + "/input \
--ImageReader.single_camera 1 \
--ImageReader.camera_model " + args.camera + " \
--SiftExtraction.use_gpu " + str(use_gpu)
exit_code = os.system(feat_extracton_cmd)
if exit_code != 0:
logging.error(f"Feature extraction failed with code {exit_code}. Exiting.")
exit(exit_code)
## Feature matching
feat_matching_cmd = colmap_command + " exhaustive_matcher \
--database_path " + args.source_path + "/distorted/database.db \
--SiftMatching.use_gpu " + str(use_gpu)
exit_code = os.system(feat_matching_cmd)
if exit_code != 0:
logging.error(f"Feature matching failed with code {exit_code}. Exiting.")
exit(exit_code)
### Bundle adjustment
# The default Mapper tolerance is unnecessarily large,
# decreasing it speeds up bundle adjustment steps.
mapper_cmd = (colmap_command + " mapper \
--database_path " + args.source_path + "/distorted/database.db \
--image_path " + args.source_path + "/input \
--output_path " + args.source_path + "/distorted/sparse \
--Mapper.ba_global_function_tolerance=0.000001")
exit_code = os.system(mapper_cmd)
if exit_code != 0:
logging.error(f"Mapper failed with code {exit_code}. Exiting.")
exit(exit_code)
if args.from_nerfstudio:
# ---------------------------------------------------------------------------------
# INGEST a scene shipped in the nerfstudio layout (this is how DL3DV ships) and write
# it out as a plain COLMAP scene with PINHOLE intrinsics.
#
# Three things have to be right or the pixels stop being comparable between methods:
#
# 1 THE INTRINSICS ARE STORED AT A DIFFERENT RESOLUTION THAN THE IMAGES. DL3DV
# records cameras at full resolution while shipping only a downscaled image
# directory, so fx, fy, cx, cy are rescaled by the true per-axis ratio first.
#
# 2 THE CAMERAS ARE OPENCV, NOT PINHOLE. 3DGS refuses OPENCV outright. The
# distortion is small but not negligible -- a few pixels of displacement at the
# corners, which is enough to move PSNR. Undistorting ONCE, here, means every
# method afterwards reads byte-identical pixels instead of two different
# undistortions.
#
# 3 THE CROP OFFSET MUST BE SUBTRACTED FROM THE PRINCIPAL POINT. getOptimalNewCamera
# Matrix(alpha=0) returns the largest all-valid rectangle, and cropping to it moves
# the optical centre by exactly the ROI offset. Forgetting that shifts the whole
# projection; it is a real and easy mistake, made by at least one published
# undistortion path.
#
# Poses and 3D points are resolution independent and are copied verbatim.
# ---------------------------------------------------------------------------------
import glob
import struct
import sys
import cv2
import numpy as np
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from scene.colmap_loader import read_intrinsics_binary
s_in = args.from_nerfstudio
col_in = os.path.join(s_in, "colmap", "sparse", "0")
imgs_in = sorted(glob.glob(os.path.join(s_in, args.images_dir, "*.png")) +
glob.glob(os.path.join(s_in, args.images_dir, "*.jpg")) +
glob.glob(os.path.join(s_in, args.images_dir, "*.JPG")))
if not imgs_in:
logging.error(f"No images in {os.path.join(s_in, args.images_dir)}")
exit(1)
os.makedirs(os.path.join(args.source_path, "images"), exist_ok=True)
os.makedirs(os.path.join(args.source_path, "sparse", "0"), exist_ok=True)
cams = read_intrinsics_binary(os.path.join(col_in, "cameras.bin"))
if len(cams) != 1:
logging.error(f"Expected 1 camera, found {len(cams)}")
exit(1)
c = list(cams.values())[0]
H, W = cv2.imread(imgs_in[0]).shape[:2]
sx, sy = W / c.width, H / c.height # (1) per-axis, never a single ratio
p = np.asarray(c.params, dtype=np.float64)
if c.model in ("OPENCV", "PINHOLE"):
fx, fy, cx, cy = p[0] * sx, p[1] * sy, p[2] * sx, p[3] * sy
dist = p[4:8].astype(np.float64) if c.model == "OPENCV" else np.zeros(4)
else:
logging.error(f"Camera model {c.model} is not supported here")
exit(1)
K = np.array([[fx, 0, cx], [0, fy, cy], [0, 0, 1]])
if np.any(dist): # (2) undistort once, on disk
Knew, roi = cv2.getOptimalNewCameraMatrix(K, dist, (W, H), 0)
Knew = Knew.copy()
Knew[0, 2] -= roi[0] # (3) the crop offset
Knew[1, 2] -= roi[1]
outW, outH = roi[2], roi[3]
else:
Knew, roi, outW, outH = K.copy(), (0, 0, W, H), W, H
print(f"Ingesting {len(imgs_in)} images: {c.model} {c.width}x{c.height} -> "
f"PINHOLE {outW}x{outH} (|k|max={np.abs(dist).max():.4f}, "
f"crop offset {roi[0]},{roi[1]})")
mapx, mapy = cv2.initUndistortRectifyMap(K, dist, None, Knew, (W, H), cv2.CV_32FC1)
for f in imgs_in:
img = cv2.remap(cv2.imread(f, cv2.IMREAD_COLOR), mapx, mapy, cv2.INTER_LINEAR)
cv2.imwrite(os.path.join(args.source_path, "images", os.path.basename(f)),
img[roi[1]:roi[1] + outH, roi[0]:roi[0] + outW],
[cv2.IMWRITE_PNG_COMPRESSION, 3])
for f in ("images.bin", "points3D.bin"):
shutil.copy(os.path.join(col_in, f),
os.path.join(args.source_path, "sparse", "0", f))
with open(os.path.join(args.source_path, "sparse", "0", "cameras.bin"), "wb") as fh:
fh.write(struct.pack("<Q", 1))
fh.write(struct.pack("<iiQQ", 1, 1, outW, outH)) # 1 = PINHOLE
fh.write(struct.pack("<4d", Knew[0, 0], Knew[1, 1], Knew[0, 2], Knew[1, 2]))
### Image undistortion
## We need to undistort our images into ideal pinhole intrinsics.
## Skipped when the scene has already been undistorted -- otherwise `--only_resize` on a
## finished COLMAP directory would go looking for an input/ that no longer needs to exist.
if not (args.only_resize or args.from_nerfstudio):
img_undist_cmd = (colmap_command + " image_undistorter \
--image_path " + args.source_path + "/input \
--input_path " + args.source_path + "/distorted/sparse/0 \
--output_path " + args.source_path + "\
--output_type COLMAP")
exit_code = os.system(img_undist_cmd)
if exit_code != 0:
logging.error(f"Mapper failed with code {exit_code}. Exiting.")
exit(exit_code)
files = os.listdir(args.source_path + "/sparse")
os.makedirs(args.source_path + "/sparse/0", exist_ok=True)
# Copy each file from the source directory to the destination directory
for file in files:
if file == '0':
continue
source_file = os.path.join(args.source_path, "sparse", file)
destination_file = os.path.join(args.source_path, "sparse", "0", file)
shutil.move(source_file, destination_file)
if(args.resize):
print("Copying and resizing...")
# Resize images.
os.makedirs(args.source_path + "/images_2", exist_ok=True)
os.makedirs(args.source_path + "/images_4", exist_ok=True)
os.makedirs(args.source_path + "/images_8", exist_ok=True)
# Get the list of files in the source directory
files = os.listdir(args.source_path + "/images")
# Copy each file from the source directory to the destination directory
for file in files:
source_file = os.path.join(args.source_path, "images", file)
destination_file = os.path.join(args.source_path, "images_2", file)
shutil.copy2(source_file, destination_file)
exit_code = os.system(magick_command + " mogrify -resize 50% " + destination_file)
if exit_code != 0:
logging.error(f"50% resize failed with code {exit_code}. Exiting.")
exit(exit_code)
destination_file = os.path.join(args.source_path, "images_4", file)
shutil.copy2(source_file, destination_file)
exit_code = os.system(magick_command + " mogrify -resize 25% " + destination_file)
if exit_code != 0:
logging.error(f"25% resize failed with code {exit_code}. Exiting.")
exit(exit_code)
destination_file = os.path.join(args.source_path, "images_8", file)
shutil.copy2(source_file, destination_file)
exit_code = os.system(magick_command + " mogrify -resize 12.5% " + destination_file)
if exit_code != 0:
logging.error(f"12.5% resize failed with code {exit_code}. Exiting.")
exit(exit_code)
if args.long_edge:
# ---------------------------------------------------------------------------------
# ONE canonical resolution, written to disk once.
#
# This is not cosmetic. If every trainer shrinks the images its own way, the ground
# truth differs between arms and the numbers stop being comparable. Doing it once,
# on disk, means every method reads byte-identical pixels.
#
# Two details that decide whether the result matches the reported numbers:
# * the interpolation is PIL's default bicubic -- the same one 3DGS applies when
# given `-r <long edge>`, so the file on disk equals what 3DGS would hold in
# memory;
# * the intrinsics are rescaled PER AXIS (sx = W/w0, sy = H/h0). Rounding W and H
# to integers makes the two ratios differ by up to ~0.27%, and using one ratio
# for both silently shifts the projection.
# ---------------------------------------------------------------------------------
import struct
import sys
from PIL import Image
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from scene.colmap_loader import read_intrinsics_binary
img_dir = os.path.join(args.source_path, "images")
cam_bin = os.path.join(args.source_path, "sparse", "0", "cameras.bin")
cam = list(read_intrinsics_binary(cam_bin).values())[0]
if cam.model != "PINHOLE":
logging.error(f"--long_edge needs PINHOLE intrinsics, got {cam.model}. "
"Run the undistortion step above first.")
exit(1)
names = sorted(os.listdir(img_dir))
w0, h0 = Image.open(os.path.join(img_dir, names[0])).size
scale = max(w0, h0) / float(args.long_edge) # max(), not w0: portrait scenes would
W, H = int(w0 / scale), int(h0 / scale) # otherwise be ENLARGED, not shrunk
print(f"Resizing {len(names)} images {w0}x{h0} -> {W}x{H} ...")
for n in names:
p = os.path.join(img_dir, n)
Image.open(p).resize((W, H)).save(p) # bicubic, as 3DGS does
fx, fy, cx, cy = cam.params
sx, sy = W / cam.width, H / cam.height
with open(cam_bin, "wb") as fh:
fh.write(struct.pack("<Q", 1))
fh.write(struct.pack("<iiQQ", 1, 1, W, H))
fh.write(struct.pack("<4d", fx * sx, fy * sy, cx * sx, cy * sy))
print(f"Rewrote cameras.bin (sx={sx:.5f}, sy={sy:.5f})")
print("Done.")