Compare commits

...
6 changed files with 611 additions and 5 deletions
+32 -1
View File
@@ -216,6 +216,37 @@ Integrate the [wan2.1-vace] video generation model to inpaint empty or newly rev
---
## Trajectory Concept
A **trajectory** in camera-comfyUI is a sequence of camera poses, each represented as a 4×4 transformation matrix. This set of matrices defines the path and orientation of the camera through 3D space, enabling smooth and complex camera movements for view synthesis, point cloud rendering, and video generation.
### Creating Trajectories
There are two main ways to create a trajectory:
- **Camera Matrices Interpolation:**
Define two or more camera poses (as matrices), and interpolate between them to generate a smooth path. The `CameraInterpolationNode` automates this process, producing a trajectory tensor for use in camera motion nodes.
- **Walking in Open3D Environment:**
Use the interactive Open3D GUI (`CameraTrajectoryNode`) to "walk" through the point cloud. As you move the camera, waypoints (poses) are recorded, forming a trajectory that can be exported and reused.
### Using Trajectories
The `CameraMotionNode` takes a trajectory (set of matrices) and interpolates camera positions and orientations along it, producing smooth camera movements for rendering sequences or videos.
---
## Point Cloud Formats
Point clouds can be saved and loaded in two formats:
- **.npy**: Numpy array format (fast, preserves all tensor data, recommended for internal pipelines).
- **.ply**: Polygon File Format (widely supported, viewable in external 3D tools).
Use the `SavePointCloud` and `LoadPointCloud` nodes to handle I/O operations in either format.
---
## Contributing
Contributions welcome! Please open issues or PRs to add features, improve docs, or refine workflows.
@@ -229,5 +260,5 @@ Contributions welcome! Please open issues or PRs to add features, improve docs,
* [x] Implement easier and more flexible camera control - more complex camera movements with more than 2 points.
* [x] Add more examples and documentation for each node.
* [x] Add pointcloud union
* [ ] Fix imports for renamed folders (e.g., inpainting_flux)
* [x] Fix imports for renamed folders (e.g., inpainting_flux)
* [x] Integrate camera movement pipeline with video models (e.g., wan2.1) for smooth, high-quality inpainting along camera trajectories.
+2 -1
View File
@@ -3,6 +3,7 @@ from .reprojection_nodes import NODE_CLASS_MAPPINGS as NCM2
from .metric_depth_nodes import NODE_CLASS_MAPPINGS as NCM3
from .flux_fisheye_filling_nodes import NODE_CLASS_MAPPINGS as NCM4
from .complex_nodes import NODE_CLASS_MAPPINGS as NCM5
NODE_CLASS_MAPPINGS = {**NCM1, **NCM2, **NCM3, **NCM4, **NCM5}
from .video_nodes import NODE_CLASS_MAPPINGS as NCM6
NODE_CLASS_MAPPINGS = {**NCM1, **NCM2, **NCM3, **NCM4, **NCM5, **NCM6}
__all__ = ["NODE_CLASS_MAPPINGS"]
Binary file not shown.
+23 -2
View File
@@ -393,6 +393,7 @@ class ProjectPointCloud:
img4 = flat.view(output_height, output_width, 4)
rgb = img4[..., :3].clamp(0, 255)
alpha = (img4[..., 3] > 0).float()
mask_init= (img4[..., 3] > 0) # initial mask
rgb *= alpha.unsqueeze(-1)
depth_img = z_front.view(output_height, output_width)
rgb_HR = rgb
@@ -406,7 +407,6 @@ class ProjectPointCloud:
idxbuf.fill_(-1)
idxbuf.scatter_reduce_(0, pix, order_m, reduce='amax', include_self=True)
win_back = idxbuf[pix] >= 0
flat.fill_(0)
flat[pix[win_back]] = colors[win_back]
back4 = flat.view(output_height, output_width, 4)
@@ -430,8 +430,29 @@ class ProjectPointCloud:
# merge only at hole locations
rgb[hole] = rgb_med[hole]
# alpha already set to 1.0 for holes
# 8 apply median blur to mask if point_size > 1 and to initial image
mask_t = alpha.unsqueeze(0).unsqueeze(0) # [1,1,H,W]
pad = point_size // 2
ksize = (point_size, point_size)
# 8) Pack and return with original script shapes
# b) grow (dilate) mask by max‑pool
mask_grow = F.max_pool2d(mask_t, kernel_size=ksize, stride=1, padding=pad)
# c) shrink (erode) by inverting, max‑pool, then inverting back
mask_shrink = 1.0 - F.max_pool2d(1.0 - mask_grow, kernel_size=ksize, stride=1, padding=pad)
# d) back to [H,W] and use as our new alpha
alpha = mask_shrink.squeeze(0).squeeze(0)
print(1)
# e) median‑filter the *whole* RGB image
# prep for kornia: [B,C,H,W]
rgb_t_full = rgb.permute(2,0,1).unsqueeze(0) # [1,3,H,W]
rgb_med_full = median_blur(rgb_t_full, ksize) # [1,3,H,W]
rgb_med_full = rgb_med_full.squeeze(0).permute(1,2,0) # [H,W,3]
# g) refill *only* the original holes with the median result
rgb[~mask_init] = rgb_med_full[~mask_init]
# 9) Pack and return with original script shapes
img = rgb.unsqueeze(0) # [1,H,W,3]
mask_out = alpha # [H,W]
depth4 = depth_img.unsqueeze(0).unsqueeze(-1) # [1,H,W,1]
+265
View File
@@ -0,0 +1,265 @@
import torch
import torch.nn.functional as F
import numpy as np
import os
import sys
from typing import Dict, Any, Tuple
from tqdm import tqdm # Added tqdm import
# Import existing pointcloud nodes and projection definitions
from .pointcloud_nodes import DepthToPointCloud, TransformPointCloud, ProjectPointCloud, Projection, PointCloudCleaner
import folder_paths
# Ensure video_depth_anything is on path
video_depth_path = os.path.join("/root/Video-Depth-Anything", "metric_depth")
if video_depth_path not in sys.path:
sys.path.append(video_depth_path)
try:
from video_depth_anything.video_depth import VideoDepthAnything
print("video_depth_anything module loaded successfully.")
except ImportError:
VideoDepthAnything = None
print("Warning: video_depth_anything module not found. Ensure it is installed correctly.")
print("error: ", sys.exc_info()[1])
class VideoCameraMotionSequence:
"""
Takes a sequence of RGB frames and corresponding depth maps,
converts each frame+depth to a pointcloud, interpolates a camera
trajectory to match video length, cleans the pointcloud if needed,
and outputs reprojected images, masks, and depth maps per frame.
"""
@classmethod
def INPUT_TYPES(cls) -> Dict[str, Any]:
return {
"required": {
# Sequence of frames: Tensor [T, H, W, 3]
"frames": ("IMAGE", {"shape_hint": [None, None, None, 3]}),
# Sequence of depth maps: Tensor [T, H, W] or [T, H, W, 1]
"depth_seq": ("TENSOR", {"shape_hint": [None, None, None]}),
# Camera trajectory waypoints: Tensor [K, 4, 4]
"trajectory": ("TENSOR", {"shape_hint": [None, 4, 4]}),
# Input projection parameters
"input_projection": (Projection.PROJECTIONS, {}),
"input_horizontal_fov": ("FLOAT", {"default": 90.0}),
"depth_scale": ("FLOAT", {"default": 1.0}),
"invert_depth": ("BOOLEAN", {"default": False}),
# Output projection parameters
"output_projection": (Projection.PROJECTIONS, {}),
"output_horizontal_fov": ("FLOAT", {"default": 90.0}),
"output_width": ("INT", {"default": 512, "min": 1}),
"output_height": ("INT", {"default": 512, "min": 1}),
"point_size": ("INT", {"default": 1, "min": 1}),
# Cleaning parameters
"voxel_size": ("FLOAT", {"default": 1.0, "min": 1e-3}),
"min_points_per_voxel": ("INT", {"default": 3, "min": 1}),
}
}
RETURN_TYPES = ("IMAGE", "MASK", "TENSOR")
RETURN_NAMES = ("video_frames", "mask_frames", "depths")
FUNCTION = "process_sequence"
CATEGORY = "Camera/Video"
def process_sequence(
self,
frames: torch.Tensor,
depth_seq: torch.Tensor,
trajectory: torch.Tensor,
input_projection: str,
input_horizontal_fov: float,
depth_scale: float,
invert_depth: bool,
output_projection: str,
output_horizontal_fov: float,
output_width: int,
output_height: int,
point_size: int,
voxel_size: float,
min_points_per_voxel: int,
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
# frames: [T, H, W, 3]
# depth_seq: [T, H, W] or [T, H, W, 1]
T, H, W, _ = frames.shape
# Interpolate trajectory to match T
K = trajectory.shape[0]
if K < 2:
interp_traj = trajectory.expand(T, 4, 4).clone()
else:
idxs = torch.linspace(0, K - 1, T, device=trajectory.device)
lower = idxs.floor().long().clamp(max=K - 2)
upper = lower + 1
alpha = (idxs - lower.float()).unsqueeze(-1).unsqueeze(-1)
traj_lower = trajectory[lower]
traj_upper = trajectory[upper]
interp_traj = traj_lower * (1 - alpha) + traj_upper * alpha
out_frames = []
out_masks = []
out_depths = []
# Add tqdm progress bar for the sequence
for frame, depth, pose in tqdm(zip(frames, depth_seq, interp_traj), total=T, desc="Processing video frames"):
if depth.dim() == 3 and depth.shape[-1] == 1:
depth = depth.squeeze(-1)
# to pointcloud
pc, = DepthToPointCloud().depth_to_pointcloud(
image=frame.permute(2, 0, 1),
input_projection=input_projection,
input_horizontal_fov=input_horizontal_fov,
depth_scale=depth_scale,
invert_depth=invert_depth,
depthmap=depth,
mask=None,
)
# optional cleaning
if min_points_per_voxel > 1:
pc, = PointCloudCleaner().clean_pointcloud(
pointcloud=pc,
width=output_width,
height=output_height,
voxel_size=voxel_size,
min_points_per_voxel=min_points_per_voxel,
)
# transform and project
pc_t, = TransformPointCloud().transform_pointcloud(pc, pose)
img_t, mask_t, depth_t = ProjectPointCloud().project_pointcloud(
pointcloud=pc_t,
output_projection=output_projection,
output_horizontal_fov=output_horizontal_fov,
output_width=output_width,
output_height=output_height,
point_size=point_size,
)
out_frames.append(img_t[0])
out_masks.append(mask_t)
out_depths.append(depth_t)
return (
torch.stack(out_frames, dim=0), # [T, 3, H, W]
torch.stack(out_masks, dim=0), # [T, H, W]
torch.stack(out_depths, dim=0), # [T, H, W]
)
class DepthFramesToVideo:
"""
Converts a sequence of depth maps into video frame tensors for saving.
"""
@classmethod
def INPUT_TYPES(cls) -> Dict[str, Any]:
return {
"required": {
"depth_seq": ("TENSOR", {"shape_hint": [None, None, None]}),
"mask_seq": ("MASK", {"shape_hint": [None, None, None]}),
"normalize": ("BOOLEAN", {"default": True}),
"invert_depth": ("BOOLEAN", {"default": False}),
}
}
RETURN_TYPES = ("TENSOR", "IMAGE")
RETURN_NAMES = ("video_frames", "depth_video")
FUNCTION = "depth_to_video_frames"
CATEGORY = "Camera/Video"
def depth_to_video_frames(
self,
depth_seq: torch.Tensor,
normalize: bool,
invert_depth: bool,
mask_seq: torch.Tensor,
) -> Tuple[torch.Tensor, torch.Tensor]:
ds = depth_seq.clone().squeeze()
if ds.dim() == 2:
ds = ds.unsqueeze(0) # [H, W] -> [1, H, W]
if ds.dim() != 3:
raise ValueError(f"Expected ds to be 3D [T, H, W], got shape {ds.shape}")
if invert_depth:
ds= 1.0 / (ds + 1e-8) # Avoid division by zero
if normalize:
# Mask: only normalize where depth > 0
mask = mask_seq>0.5
if mask.any():
#percentile first 10 percent min
# sample
minv = ds[mask]
# sample 10000 and find 10% quantile
if minv.numel() > 10000:
minv = minv[torch.randperm(minv.numel())[:10000]]
minv = minv.quantile(0.2)
minv = minv if minv > 0.1 else 0.1 # Avoid division by zero
#percentile last 10 percent max
maxv = ds[mask]
if maxv.numel() > 10000:
maxv = maxv[torch.randperm(maxv.numel())[:10000]]
maxv = maxv.quantile(0.98)
maxv = maxv if maxv < 100 else 100
print(f"Normalizing depth: min={minv}, max={maxv}")
ds_norm = (ds - minv) / (maxv - minv + 1e-8)
ds = ds_norm.clamp(0, 1) # torch.where(mask, ds_norm, ds) # Only normalize valid values
else:
print("Warning: No valid depth values for normalization.")
# expand to 3 channels: [T, H, W] -> [T, 3, H, W]
raw = depth_seq.clone().squeeze()
ds_u8 = (ds * 255.0).round().to(torch.uint8)
raw_u8 = (raw.clamp(0, 255)).to(torch.uint8) # if raw is already in a displayable range
# expand to 3 channels and permute to HWC
ds_color = ds_u8.unsqueeze(1).repeat(1, 3, 1, 1).permute(0, 2, 3, 1)
raw_color = raw_u8.unsqueeze(1).repeat(1, 3, 1, 1).permute(0, 2, 3, 1)
return raw_color, ds_color # [T, 3, H, W] -> [T, H, W, 3]
class VideoMetricDepthEstimate:
"""
Estimates metric depth for a sequence of frames using VideoDepthAnything.
"""
@classmethod
def INPUT_TYPES(cls) -> Dict[str, Any]:
# model files (.pth) in input directory
model_dir = os.path.join(os.getcwd(), "models", "checkpoints")
os.makedirs(model_dir, exist_ok=True)
files = [f for f in os.listdir(model_dir) if f.lower().endswith(('.pth', '.ckpt', '.safetensors'))]
return {
"required": {
"frames": ("IMAGE", {"shape_hint": [None, None, None, 3]}),
"model_checkpoint": (files, {"file_chooser": True}),
"input_size": ("INT", {"default": 518, "min": 64, "max": 2048}),
"max_fps": ("INT", {"default": 60, "min": 1}),
}
}
RETURN_TYPES = ("TENSOR", "FLOAT")
RETURN_NAMES = ("metric_depths", "fps")
FUNCTION = "estimate_metric_depth"
CATEGORY = "Camera/Video"
def estimate_metric_depth(
self,
frames: torch.Tensor,
model_checkpoint: str,
input_size: int,
max_fps: int,
) -> Tuple[torch.Tensor, float]:
if VideoDepthAnything is None:
raise ImportError("VideoDepthAnything library not found")
device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
# if max input<1.5 normalize to 0-255
if frames.max() < 1.5:
frames = (frames * 255)
model = VideoDepthAnything(**{"encoder": "vitl", "features": 256, "out_channels": [256,512,1024,1024]})
state = torch.load("/root/ComfyUI/models/checkpoints/{}".format(model_checkpoint), map_location='cpu')
model.load_state_dict(state, strict=True)
model = model.to(device).eval()
np_frames = frames.cpu().numpy().astype(np.uint8)
metric_depths, fps = model.infer_video_depth(np_frames, max_fps, input_size=input_size, device=device.type, fp32=False)
return (torch.from_numpy(metric_depths), float(fps))
# Register nodes
NODE_CLASS_MAPPINGS = {
"VideoCameraMotionSequence": VideoCameraMotionSequence,
"VideoMetricDepthEstimate": VideoMetricDepthEstimate,
"DepthFramesToVideo": DepthFramesToVideo,
}
+289 -1
View File
@@ -1 +1,289 @@
{"id":"dd56c0bf-7405-406e-924f-42b2feacb73f","revision":0,"last_node_id":6,"last_link_id":4,"nodes":[{"id":1,"type":"TransformToMatrix","pos":[-337.9580993652344,1500.5211181640625],"size":[315,154],"flags":{},"order":0,"mode":0,"inputs":[],"outputs":[{"localized_name":"MAT_4X4","name":"MAT_4X4","type":"MAT_4X4","links":[1]}],"properties":{"Node name for S&R":"TransformToMatrix"},"widgets_values":[0,0,0,0,0]},{"id":2,"type":"CameraMotion","pos":[130.0218505859375,1516.7567138671875],"size":[367.79998779296875,218],"flags":{},"order":3,"mode":0,"inputs":[{"localized_name":"pointcloud","name":"pointcloud","type":"TENSOR","link":4},{"localized_name":"initial_matrix","name":"initial_matrix","type":"MAT_4X4","link":1},{"localized_name":"final_matrix","name":"final_matrix","type":"MAT_4X4","link":2}],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[3]}],"properties":{"Node name for S&R":"CameraMotion"},"widgets_values":[24,"PINHOLE",90,1024,1024,2]},{"id":3,"type":"SaveWEBM","pos":[606.5880737304688,1515.5966796875],"size":[315,437],"flags":{},"order":4,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":3}],"outputs":[],"properties":{},"widgets_values":["ComfyUI","vp9",10.000000000000002,32]},{"id":5,"type":"TransformToMatrix","pos":[-262.9134521484375,1740.8876953125],"size":[315,154],"flags":{},"order":1,"mode":0,"inputs":[],"outputs":[{"localized_name":"MAT_4X4","name":"MAT_4X4","type":"MAT_4X4","links":[2]}],"properties":{"Node name for S&R":"TransformToMatrix"},"widgets_values":[0.10000000000000002,0,0,0,0]},{"id":6,"type":"LoadPointCloud","pos":[-357.24761962890625,1294.427734375],"size":[315,58],"flags":{},"order":2,"mode":0,"inputs":[],"outputs":[{"localized_name":"TENSOR","name":"TENSOR","type":"TENSOR","links":[4]}],"properties":{"Node name for S&R":"LoadPointCloud"},"widgets_values":["ComfyUIPointCloud_00001.ply"]}],"links":[[1,1,0,2,1,"MAT_4X4"],[2,5,0,2,2,"MAT_4X4"],[3,2,0,3,0,"IMAGE"],[4,6,0,2,0,"TENSOR"]],"groups":[],"config":{},"extra":{"ds":{"scale":1.351305709310409,"offset":[-187.6257577580669,-1567.2143321744395]}},"version":0.4}
{
"id": "dd56c0bf-7405-406e-924f-42b2feacb73f",
"revision": 0,
"last_node_id": 8,
"last_link_id": 9,
"nodes": [
{
"id": 3,
"type": "SaveWEBM",
"pos": [
606.5880737304688,
1515.5966796875
],
"size": [
315,
437
],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 5
}
],
"outputs": [],
"properties": {},
"widgets_values": [
"ComfyUI",
"vp9",
10.000000000000002,
32
]
},
{
"id": 7,
"type": "CameraMotionNode",
"pos": [
176.07933044433594,
1504.22705078125
],
"size": [
278.75,
270
],
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "pointcloud",
"type": "TENSOR",
"link": 6
},
{
"name": "trajectory",
"type": "TENSOR",
"link": 9
}
],
"outputs": [
{
"name": "motion_frames",
"type": "IMAGE",
"links": [
5
]
},
{
"name": "mask_frames",
"type": "MASK",
"links": null
}
],
"properties": {
"Node name for S&R": "CameraMotionNode"
},
"widgets_values": [
10,
"PINHOLE",
90,
512,
512,
1,
0,
false,
false
]
},
{
"id": 6,
"type": "LoadPointCloud",
"pos": [
-357.24761962890625,
1294.427734375
],
"size": [
315,
58
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "loaded pointcloud",
"type": "TENSOR",
"links": [
6
]
}
],
"properties": {
"Node name for S&R": "LoadPointCloud"
},
"widgets_values": [
"ComfyUIPointCloud_00001.ply"
]
},
{
"id": 1,
"type": "TransformToMatrix",
"pos": [
-537.2319946289062,
1491.40380859375
],
"size": [
315,
154
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "transformation matrix",
"type": "MAT_4X4",
"links": [
7
]
}
],
"properties": {
"Node name for S&R": "TransformToMatrix"
},
"widgets_values": [
0,
0,
0,
0,
0
]
},
{
"id": 5,
"type": "TransformToMatrix",
"pos": [
-531.216796875,
1701.1632080078125
],
"size": [
315,
154
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "transformation matrix",
"type": "MAT_4X4",
"links": [
8
]
}
],
"properties": {
"Node name for S&R": "TransformToMatrix"
},
"widgets_values": [
0.10000000000000002,
0,
0,
0,
0
]
},
{
"id": 8,
"type": "CameraInterpolationNode",
"pos": [
-121.91971588134766,
1597.3514404296875
],
"size": [
200.21640014648438,
46
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [
{
"name": "initial_matrix",
"type": "MAT_4X4",
"link": 7
},
{
"name": "final_matrix",
"type": "MAT_4X4",
"link": 8
}
],
"outputs": [
{
"name": "trajectory",
"type": "TENSOR",
"links": [
9
]
}
],
"properties": {
"Node name for S&R": "CameraInterpolationNode"
},
"widgets_values": []
}
],
"links": [
[
5,
7,
0,
3,
0,
"IMAGE"
],
[
6,
6,
0,
7,
0,
"TENSOR"
],
[
7,
1,
0,
8,
0,
"MAT_4X4"
],
[
8,
5,
0,
8,
1,
"MAT_4X4"
],
[
9,
8,
0,
7,
1,
"TENSOR"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 1.015255979947716,
"offset": [
636.7531305750655,
-1186.3099424359816
]
},
"frontendVersion": "1.21.7"
},
"version": 0.4
}