Initial commit

Co-authored-by: Jensen Zhou <jensen.zhou@stability.ai>
Co-authored-by: Aaryaman Vasishta <aaryaman.vasishta@stability.ai>
This commit is contained in:
Hang Gao
2025-03-18 07:49:36 -07:00
co-authored by Jensen Zhou Aaryaman Vasishta
commit 8a640b7316
53 changed files with 8131 additions and 0 deletions
+156
View File
@@ -0,0 +1,156 @@
# :bar_chart: Benchmark
We provide <a href="https://github.com/Stability-AI/stable-virtual-camera/releases/tag/benchmark">in this release</a> (`benchmark.zip`) with the following 17 entries as a benchmark to evaluate NVS models.
We hope this will help standardize the evaluation of NVS models and facilitate fair comparison between different methods.
<table>
<thead>
<tr>
<th align="center">Dataset</th>
<th align="center">Split</th>
<th align="center">Path</th>
<th align="center">Content</th>
<th align="center">Image Preprocessing</th>
<th align="center">Image Postprocessing</th>
</tr>
</thead>
<tbody>
<tr>
<td align="center">OmniObject3D</td>
<td align="center"><code>S</code> (SV3D), <code>O</code> (Ours) </td>
<td align="center"><code>omniobject3d</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
<tr>
<td align="center">GSO</td>
<td align="center"><code>S</code> (SV3D), <code>O</code> (Ours) </td>
<td align="center"><code>gso</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
<tr>
<td align="center" rowspan="4">RealEstate10K</td>
<td align="center"><code>D</code> (4DiM) </td>
<td align="center"><code>re10k-4dim</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">resize to 256</td>
</tr>
<tr>
<td align="center"><code>R</code> (ReconFusion) </td>
<td align="center"><code>re10k</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
<tr>
<td align="center"><code>P</code> (pixelSplat) </td>
<td align="center"><code>re10k-pixelsplat</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">resize to 256</td>
</tr>
<tr>
<td align="center"><code>V</code> (ViewCrafter) </td>
<td align="center"><code>re10k-viewcrafter</code></td>
<td align="center"><code>images/*.png</code>,<code>transforms.json</code>,<code>train_test_split_*.json</code></td>
<td align="center">resize the shortest side to 576 (<code>--L_short 576</code>)</td>
<td align="center">center crop</td>
</tr>
<tr>
<td align="center">LLFF</td>
<td align="center"><code>R</code> (ReconFusion) </td>
<td align="center"><code>llff</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
<tr>
<td align="center">DTU</td>
<td align="center"><code>R</code> (ReconFusion) </td>
<td align="center"><code>dtu</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
<tr>
<td align="center" rowspan="2">CO3D</td>
<td align="center"><code>R</code> (ReconFusion) </td>
<td align="center"><code>co3d</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
<tr>
<td align="center"><code>V</code> (ViewCrafter) </td>
<td align="center"><code>co3d-viewcrafter</code></td>
<td align="center"><code>images/*.png</code>,<code>transforms.json</code>,<code>train_test_split_*.json</code></td>
<td align="center">resize the shortest side to 576 (<code>--L_short 576</code>)</td>
<td align="center">center crop</td>
</tr>
<tr>
<td align="center" rowspan="2" >WildRGB-D</td>
<td align="center"><code>Oₑ</code> (Ours, easy) </td>
<td align="center"><code>wildgbd/easy</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
<tr>
<td align="center"><code>Oₕ</code> (Ours, hard) </td>
<td align="center"><code>wildgbd/hard</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
<tr>
<td align="center">Mip-NeRF360</td>
<td align="center"><code>R</code> (ReconFusion) </td>
<td align="center"><code>mipnerf360</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
<tr>
<td align="center" rowspan="2">DL3DV-140</td>
<td align="center"><code>O</code> (Ours) </td>
<td align="center"><code>dl3dv10</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
<tr>
<td align="center"><code>L</code> (Long-LRM) </td>
<td align="center"><code>dl3dv140</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
<tr>
<td align="center" rowspan="2">Tanks and Temples</td>
<td align="center"><code>V</code> (ViewCrafter) </td>
<td align="center"><code>tnt-viewcrafter</code></td>
<td align="center"><code>images/*.png</code>,<code>transforms.json</code>,<code>train_test_split_*.json</code></td>
<td align="center">resize the shortest side to 576 (<code>--L_short 576</code>)</td>
<td align="center">center crop</td>
</tr>
<tr>
<td align="center"><code>L</code> (Long-LRM) </td>
<td align="center"><code>tnt-longlrm</code></td>
<td align="center"><code>train_test_split_*.json</code></td>
<td align="center">center crop to 576</td>
<td align="center">\</td>
</tr>
</tbody>
</table>
- For entries without `images/*.png` and `transforms.json`, we use the images from the original dataset after converting them into the `reconfusion` format, which is then parsable by `ReconfusionParser` (`seva/data_io.py`).
Please note that during this conversion, you should sort the images by `sorted(image_paths)`, which is then directly indexable by our train/test ids. We provide in `benchmark/export_reconfusion_example.py` an example script converting an existing academic dataset into the the scene folders.
- For evaluation and benchmarking, we first conduct operations in the `Image Preprocessing` column to the model input and then operations in the `Image Postprocessing` column to the model output. The final processed samples are used for metric computation.
## Acknowledgment
We would like to thank Wangbo Yu, Aleksander Hołyński, Saurabh Saxena, and Ziwen Chen for their kind clarification on experiment settings.
+137
View File
@@ -0,0 +1,137 @@
import argparse
import json
import os
import numpy as np
from PIL import Image
try:
from sklearn.cluster import KMeans # type: ignore[import]
except ImportError:
print("Please install sklearn to use this script.")
exit(1)
# Define the folder containing the image and JSON files
subfolder = "/path/to/your/dataset"
output_file = os.path.join(subfolder, "transforms.json")
# List to hold the frames
frames = []
# Iterate over the files in the folder
for file in sorted(os.listdir(subfolder)):
if file.endswith(".json"):
# Read the JSON file containing camera extrinsics and intrinsics
json_path = os.path.join(subfolder, file)
with open(json_path, "r") as f:
data = json.load(f)
# Read the corresponding image file
image_file = file.replace(".json", ".png")
image_path = os.path.join(subfolder, image_file)
if not os.path.exists(image_path):
print(f"Image file not found for {file}, skipping...")
continue
with Image.open(image_path) as img:
w, h = img.size
# Extract and normalize intrinsic matrix K
K = data["K"]
fx = K[0][0] * w
fy = K[1][1] * h
cx = K[0][2] * w
cy = K[1][2] * h
# Extract the transformation matrix
transform_matrix = np.array(data["c2w"])
# Adjust for OpenGL convention
transform_matrix[..., [1, 2]] *= -1
# Add the frame data
frames.append(
{
"fl_x": fx,
"fl_y": fy,
"cx": cx,
"cy": cy,
"w": w,
"h": h,
"file_path": f"./{os.path.relpath(image_path, subfolder)}",
"transform_matrix": transform_matrix.tolist(),
}
)
# Create the output dictionary
transforms_data = {"orientation_override": "none", "frames": frames}
# Write to the transforms.json file
with open(output_file, "w") as f:
json.dump(transforms_data, f, indent=4)
print(f"transforms.json generated at {output_file}")
# Train-test split function using K-means clustering with stride
def create_train_test_split(frames, n, output_path, stride):
# Prepare the data for K-means
positions = []
for frame in frames:
transform_matrix = np.array(frame["transform_matrix"])
position = transform_matrix[:3, 3] # 3D camera position
direction = transform_matrix[:3, 2] / np.linalg.norm(
transform_matrix[:3, 2]
) # Normalized 3D direction
positions.append(np.concatenate([position, direction]))
positions = np.array(positions)
# Apply K-means clustering
kmeans = KMeans(n_clusters=n, random_state=42)
kmeans.fit(positions)
centers = kmeans.cluster_centers_
# Find the index closest to each cluster center
train_ids = []
for center in centers:
distances = np.linalg.norm(positions - center, axis=1)
train_ids.append(int(np.argmin(distances))) # Convert to Python int
# Remaining indices as test_ids, applying stride
all_indices = set(range(len(frames)))
remaining_indices = sorted(all_indices - set(train_ids))
test_ids = [
int(idx) for idx in remaining_indices[::stride]
] # Convert to Python int
# Create the split data
split_data = {"train_ids": sorted(train_ids), "test_ids": test_ids}
with open(output_path, "w") as f:
json.dump(split_data, f, indent=4)
print(f"Train-test split file generated at {output_path}")
# Parse arguments
if __name__ == "__main__":
parser = argparse.ArgumentParser(
description="Generate train-test split JSON file using K-means clustering."
)
parser.add_argument(
"--n",
type=int,
required=True,
help="Number of frames to include in the training set.",
)
parser.add_argument(
"--stride",
type=int,
default=1,
help="Stride for selecting test frames (not used with K-means).",
)
args = parser.parse_args()
# Create train-test split
train_test_split_path = os.path.join(subfolder, f"train_test_split_{args.n}.json")
create_train_test_split(frames, args.n, train_test_split_path, args.stride)