Initial commit
Co-authored-by: Jensen Zhou <jensen.zhou@stability.ai> Co-authored-by: Aaryaman Vasishta <aaryaman.vasishta@stability.ai>
This commit is contained in:
@@ -0,0 +1,156 @@
|
||||
# :bar_chart: Benchmark
|
||||
|
||||
We provide <a href="https://github.com/Stability-AI/stable-virtual-camera/releases/tag/benchmark">in this release</a> (`benchmark.zip`) with the following 17 entries as a benchmark to evaluate NVS models.
|
||||
We hope this will help standardize the evaluation of NVS models and facilitate fair comparison between different methods.
|
||||
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th align="center">Dataset</th>
|
||||
<th align="center">Split</th>
|
||||
<th align="center">Path</th>
|
||||
<th align="center">Content</th>
|
||||
<th align="center">Image Preprocessing</th>
|
||||
<th align="center">Image Postprocessing</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
<tr>
|
||||
<td align="center">OmniObject3D</td>
|
||||
<td align="center"><code>S</code> (SV3D), <code>O</code> (Ours) </td>
|
||||
<td align="center"><code>omniobject3d</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center">GSO</td>
|
||||
<td align="center"><code>S</code> (SV3D), <code>O</code> (Ours) </td>
|
||||
<td align="center"><code>gso</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" rowspan="4">RealEstate10K</td>
|
||||
<td align="center"><code>D</code> (4DiM) </td>
|
||||
<td align="center"><code>re10k-4dim</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">resize to 256</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center"><code>R</code> (ReconFusion) </td>
|
||||
<td align="center"><code>re10k</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center"><code>P</code> (pixelSplat) </td>
|
||||
<td align="center"><code>re10k-pixelsplat</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">resize to 256</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center"><code>V</code> (ViewCrafter) </td>
|
||||
<td align="center"><code>re10k-viewcrafter</code></td>
|
||||
<td align="center"><code>images/*.png</code>,<code>transforms.json</code>,<code>train_test_split_*.json</code></td>
|
||||
<td align="center">resize the shortest side to 576 (<code>--L_short 576</code>)</td>
|
||||
<td align="center">center crop</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center">LLFF</td>
|
||||
<td align="center"><code>R</code> (ReconFusion) </td>
|
||||
<td align="center"><code>llff</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center">DTU</td>
|
||||
<td align="center"><code>R</code> (ReconFusion) </td>
|
||||
<td align="center"><code>dtu</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" rowspan="2">CO3D</td>
|
||||
<td align="center"><code>R</code> (ReconFusion) </td>
|
||||
<td align="center"><code>co3d</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center"><code>V</code> (ViewCrafter) </td>
|
||||
<td align="center"><code>co3d-viewcrafter</code></td>
|
||||
<td align="center"><code>images/*.png</code>,<code>transforms.json</code>,<code>train_test_split_*.json</code></td>
|
||||
<td align="center">resize the shortest side to 576 (<code>--L_short 576</code>)</td>
|
||||
<td align="center">center crop</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" rowspan="2" >WildRGB-D</td>
|
||||
<td align="center"><code>Oₑ</code> (Ours, easy) </td>
|
||||
<td align="center"><code>wildgbd/easy</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center"><code>Oₕ</code> (Ours, hard) </td>
|
||||
<td align="center"><code>wildgbd/hard</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center">Mip-NeRF360</td>
|
||||
<td align="center"><code>R</code> (ReconFusion) </td>
|
||||
<td align="center"><code>mipnerf360</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" rowspan="2">DL3DV-140</td>
|
||||
<td align="center"><code>O</code> (Ours) </td>
|
||||
<td align="center"><code>dl3dv10</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center"><code>L</code> (Long-LRM) </td>
|
||||
<td align="center"><code>dl3dv140</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" rowspan="2">Tanks and Temples</td>
|
||||
<td align="center"><code>V</code> (ViewCrafter) </td>
|
||||
<td align="center"><code>tnt-viewcrafter</code></td>
|
||||
<td align="center"><code>images/*.png</code>,<code>transforms.json</code>,<code>train_test_split_*.json</code></td>
|
||||
<td align="center">resize the shortest side to 576 (<code>--L_short 576</code>)</td>
|
||||
<td align="center">center crop</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center"><code>L</code> (Long-LRM) </td>
|
||||
<td align="center"><code>tnt-longlrm</code></td>
|
||||
<td align="center"><code>train_test_split_*.json</code></td>
|
||||
<td align="center">center crop to 576</td>
|
||||
<td align="center">\</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
|
||||
- For entries without `images/*.png` and `transforms.json`, we use the images from the original dataset after converting them into the `reconfusion` format, which is then parsable by `ReconfusionParser` (`seva/data_io.py`).
|
||||
Please note that during this conversion, you should sort the images by `sorted(image_paths)`, which is then directly indexable by our train/test ids. We provide in `benchmark/export_reconfusion_example.py` an example script converting an existing academic dataset into the the scene folders.
|
||||
- For evaluation and benchmarking, we first conduct operations in the `Image Preprocessing` column to the model input and then operations in the `Image Postprocessing` column to the model output. The final processed samples are used for metric computation.
|
||||
|
||||
## Acknowledgment
|
||||
|
||||
We would like to thank Wangbo Yu, Aleksander Hołyński, Saurabh Saxena, and Ziwen Chen for their kind clarification on experiment settings.
|
||||
@@ -0,0 +1,137 @@
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
try:
|
||||
from sklearn.cluster import KMeans # type: ignore[import]
|
||||
except ImportError:
|
||||
print("Please install sklearn to use this script.")
|
||||
exit(1)
|
||||
|
||||
# Define the folder containing the image and JSON files
|
||||
subfolder = "/path/to/your/dataset"
|
||||
output_file = os.path.join(subfolder, "transforms.json")
|
||||
|
||||
# List to hold the frames
|
||||
frames = []
|
||||
|
||||
# Iterate over the files in the folder
|
||||
for file in sorted(os.listdir(subfolder)):
|
||||
if file.endswith(".json"):
|
||||
# Read the JSON file containing camera extrinsics and intrinsics
|
||||
json_path = os.path.join(subfolder, file)
|
||||
with open(json_path, "r") as f:
|
||||
data = json.load(f)
|
||||
|
||||
# Read the corresponding image file
|
||||
image_file = file.replace(".json", ".png")
|
||||
image_path = os.path.join(subfolder, image_file)
|
||||
if not os.path.exists(image_path):
|
||||
print(f"Image file not found for {file}, skipping...")
|
||||
continue
|
||||
with Image.open(image_path) as img:
|
||||
w, h = img.size
|
||||
|
||||
# Extract and normalize intrinsic matrix K
|
||||
K = data["K"]
|
||||
fx = K[0][0] * w
|
||||
fy = K[1][1] * h
|
||||
cx = K[0][2] * w
|
||||
cy = K[1][2] * h
|
||||
|
||||
# Extract the transformation matrix
|
||||
transform_matrix = np.array(data["c2w"])
|
||||
# Adjust for OpenGL convention
|
||||
transform_matrix[..., [1, 2]] *= -1
|
||||
|
||||
# Add the frame data
|
||||
frames.append(
|
||||
{
|
||||
"fl_x": fx,
|
||||
"fl_y": fy,
|
||||
"cx": cx,
|
||||
"cy": cy,
|
||||
"w": w,
|
||||
"h": h,
|
||||
"file_path": f"./{os.path.relpath(image_path, subfolder)}",
|
||||
"transform_matrix": transform_matrix.tolist(),
|
||||
}
|
||||
)
|
||||
|
||||
# Create the output dictionary
|
||||
transforms_data = {"orientation_override": "none", "frames": frames}
|
||||
|
||||
# Write to the transforms.json file
|
||||
with open(output_file, "w") as f:
|
||||
json.dump(transforms_data, f, indent=4)
|
||||
|
||||
print(f"transforms.json generated at {output_file}")
|
||||
|
||||
|
||||
# Train-test split function using K-means clustering with stride
|
||||
def create_train_test_split(frames, n, output_path, stride):
|
||||
# Prepare the data for K-means
|
||||
positions = []
|
||||
for frame in frames:
|
||||
transform_matrix = np.array(frame["transform_matrix"])
|
||||
position = transform_matrix[:3, 3] # 3D camera position
|
||||
direction = transform_matrix[:3, 2] / np.linalg.norm(
|
||||
transform_matrix[:3, 2]
|
||||
) # Normalized 3D direction
|
||||
positions.append(np.concatenate([position, direction]))
|
||||
|
||||
positions = np.array(positions)
|
||||
|
||||
# Apply K-means clustering
|
||||
kmeans = KMeans(n_clusters=n, random_state=42)
|
||||
kmeans.fit(positions)
|
||||
centers = kmeans.cluster_centers_
|
||||
|
||||
# Find the index closest to each cluster center
|
||||
train_ids = []
|
||||
for center in centers:
|
||||
distances = np.linalg.norm(positions - center, axis=1)
|
||||
train_ids.append(int(np.argmin(distances))) # Convert to Python int
|
||||
|
||||
# Remaining indices as test_ids, applying stride
|
||||
all_indices = set(range(len(frames)))
|
||||
remaining_indices = sorted(all_indices - set(train_ids))
|
||||
test_ids = [
|
||||
int(idx) for idx in remaining_indices[::stride]
|
||||
] # Convert to Python int
|
||||
|
||||
# Create the split data
|
||||
split_data = {"train_ids": sorted(train_ids), "test_ids": test_ids}
|
||||
|
||||
with open(output_path, "w") as f:
|
||||
json.dump(split_data, f, indent=4)
|
||||
|
||||
print(f"Train-test split file generated at {output_path}")
|
||||
|
||||
|
||||
# Parse arguments
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Generate train-test split JSON file using K-means clustering."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--n",
|
||||
type=int,
|
||||
required=True,
|
||||
help="Number of frames to include in the training set.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--stride",
|
||||
type=int,
|
||||
default=1,
|
||||
help="Stride for selecting test frames (not used with K-means).",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# Create train-test split
|
||||
train_test_split_path = os.path.join(subfolder, f"train_test_split_{args.n}.json")
|
||||
create_train_test_split(frames, args.n, train_test_split_path, args.stride)
|
||||
Reference in New Issue
Block a user