Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
012ab737d6 | ||
|
|
2f9aa76478 | ||
|
|
918c882abd | ||
|
|
79a1c2d6f3 | ||
|
|
d77af07189 | ||
|
|
6f8ed9d382 | ||
|
|
d85c2125d1 | ||
|
|
358b2b55ae | ||
|
|
8598a7b51c | ||
|
|
90df387c2a | ||
|
|
74d71a91f2 | ||
|
|
55820b53e8 | ||
|
|
f6b6707dd8 | ||
|
|
49f9700880 | ||
|
|
b961c2988d | ||
|
|
2b1b716495 | ||
|
|
839fa5883a | ||
|
|
74f98be5c2 | ||
|
|
c1e1b55464 | ||
|
|
08b48b2fcb | ||
|
|
4812973e0f | ||
|
|
57fc1e8593 | ||
|
|
abdfdc188e | ||
|
|
ca6f904fc3 | ||
|
|
696a4ad763 | ||
|
|
d67cd61c3c | ||
|
|
17a0edb3c1 | ||
|
|
5592382b44 | ||
|
|
de679db043 | ||
|
|
bc358f8263 | ||
|
|
1d0ae6dc61 | ||
|
|
3bedb49949 | ||
|
|
b67ae9a0a2 | ||
|
|
1d072713b4 | ||
|
|
8b331c6422 | ||
|
|
90bcfd720f | ||
|
|
9c02934b76 | ||
|
|
e8c4642104 | ||
|
|
f375b69bf0 | ||
|
|
892cfa474a | ||
|
|
d604ec8646 | ||
|
|
6f0fe189cd | ||
|
|
c6ab9e2cbb | ||
|
|
a32489234a | ||
|
|
38d18f19b2 | ||
|
|
c52390afa8 | ||
|
|
4771b9ee45 | ||
|
|
962027f75a | ||
|
|
65b7671bca | ||
|
|
296175de54 |
@@ -0,0 +1,10 @@
|
||||
# Excluded from the ComfyUI Registry archive (not from git).
|
||||
demo_images/
|
||||
notebooks/
|
||||
docs/
|
||||
screenshot1.ply
|
||||
__pycache__/
|
||||
models/
|
||||
.github/
|
||||
Makefile
|
||||
install.sh
|
||||
@@ -0,0 +1,28 @@
|
||||
name: Publish to Comfy registry
|
||||
on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- "pyproject.toml"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
publish-node:
|
||||
name: Publish Custom Node to registry
|
||||
runs-on: ubuntu-latest
|
||||
if: ${{ github.repository_owner == 'Alexankharin' }}
|
||||
steps:
|
||||
- name: Check out code
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
# The SHARP submodule must be materialized so [tool.comfy].includes
|
||||
# can pack it into the published archive.
|
||||
submodules: recursive
|
||||
- name: Publish Custom Node
|
||||
uses: Comfy-Org/publish-node-action@main
|
||||
with:
|
||||
personal_access_token: ${{ secrets.REGISTRY_ACCESS_TOKEN }}
|
||||
@@ -0,0 +1,34 @@
|
||||
name: Validate workflows & smoke tests
|
||||
|
||||
# Guards against node-schema drift silently breaking the shipped workflows:
|
||||
# validate_workflows.py compares every stored widgets_values against the
|
||||
# current INPUT_TYPES (see docs/workflows_review.md).
|
||||
on:
|
||||
pull_request:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
validate:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check out code
|
||||
uses: actions/checkout@v4
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
- name: Install CPU test dependencies
|
||||
run: |
|
||||
pip install torch torchvision --index-url https://download.pytorch.org/whl/cpu
|
||||
pip install numpy pillow scipy tqdm
|
||||
- name: Workflow schema validation
|
||||
run: python notebooks/validate_workflows.py
|
||||
- name: Installer logic tests
|
||||
run: python notebooks/test_install_logic.py
|
||||
- name: 4D node smoke tests
|
||||
run: python notebooks/smoke_test_4d.py
|
||||
@@ -0,0 +1,2 @@
|
||||
__pycache__/
|
||||
*.pyc
|
||||
@@ -0,0 +1,3 @@
|
||||
[submodule "submodules/ml-sharpt"]
|
||||
path = submodules/ml-sharpt
|
||||
url = https://github.com/apple/ml-sharp
|
||||
@@ -0,0 +1,29 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Alexander Kharin
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
---
|
||||
|
||||
Note: the bundled directory `submodules/ml-sharpt` contains Apple's ml-sharp
|
||||
project and is licensed separately under the terms in
|
||||
`submodules/ml-sharpt/LICENSE` (source) and `submodules/ml-sharpt/LICENSE_MODEL`
|
||||
(model weights, research-only). The MIT license above does not apply to that
|
||||
directory.
|
||||
@@ -0,0 +1,19 @@
|
||||
.PHONY: install install_all install_modules download_flux download_vae
|
||||
|
||||
# install everything except WAN‑VACE downloads
|
||||
install:
|
||||
./install.sh install
|
||||
|
||||
# install everything + WAN‑VACE + HF login
|
||||
install_all:
|
||||
./install.sh all
|
||||
|
||||
# lower‑level helpers
|
||||
install_modules:
|
||||
./install.sh modules
|
||||
|
||||
download_flux:
|
||||
./install.sh flux
|
||||
|
||||
download_vae:
|
||||
./install.sh vae
|
||||
@@ -1,6 +1,7 @@
|
||||
# camera-comfyUI
|
||||
|
||||
[](https://deepwiki.com/Alexankharin/camera-comfyUI)
|
||||

|
||||

|
||||
|
||||
> Custom ComfyUI nodes for advanced reprojections, point cloud processing, and camera-driven workflows.
|
||||
|
||||
@@ -13,6 +14,7 @@
|
||||
* [Installation](#installation)
|
||||
* [Node Categories](#node-categories)
|
||||
* [Node Reference](#node-reference)
|
||||
* [Video → 4D World](#video--4d-world)
|
||||
* [Workflows](#workflows)
|
||||
* [Example Workflows](#example-workflows)
|
||||
* [Contributing](#contributing)
|
||||
@@ -34,6 +36,24 @@ A collection of ComfyUI custom nodes to handle diverse camera projections (pinho
|
||||
|
||||
## Installation
|
||||
|
||||
### Option A — ComfyUI Manager (recommended)
|
||||
|
||||
The node pack is published to the [ComfyUI Registry](https://registry.comfy.org) as **`camera-comfyui`** (publisher `alexk`). In ComfyUI, open **Manager → Custom Nodes Manager**, search for **camera-comfyUI**, and click **Install**, then restart ComfyUI.
|
||||
|
||||
Installation is fully automatic: ComfyUI-Manager installs `requirements.txt` and then runs this pack's `install.py`, which sets up everything the optional nodes need — no manual steps:
|
||||
|
||||
* **vggt** (`VideoPoseEstimator`) — pip-installed from GitHub over https (it is not on PyPI).
|
||||
* **SHARP** (`ImageToSplat`, `VideoToFusedSplats`, …) — the `submodules/ml-sharpt` checkout is bundled in the registry package (and fetched via `git submodule`/clone for git installs), and its Python deps come from `requirements.txt`.
|
||||
* **gsplat** — pip-installed; its CUDA kernels JIT-compile on first use.
|
||||
* **ComfyUI-Flux-Inpainting** (`OutpaintAnyProjection`, `SplatTrajectoryEnricher`) — cloned automatically into `custom_nodes/inpainting_flux` (skipped if you already have the pack under any of its usual folder names).
|
||||
* **Example inputs** — the sample image/trajectory files referenced by the bundled workflows are copied into ComfyUI's `input/` folder, so the templates run immediately.
|
||||
|
||||
Each step is optional and non-fatal: if one fails (e.g. no network), only the nodes that need it stay disabled — re-run `python install.py` inside the pack folder to retry.
|
||||
|
||||
> **Maintainers:** releases are automated — bumping `version` in `pyproject.toml` on `main` triggers `.github/workflows/publish_action.yml`, which publishes the new version to the registry (requires the `REGISTRY_ACCESS_TOKEN` repo secret).
|
||||
|
||||
### Option B — Manual install (git)
|
||||
|
||||
1. **Clone** into your ComfyUI custom nodes folder:
|
||||
|
||||
```bash
|
||||
@@ -46,28 +66,29 @@ A collection of ComfyUI custom nodes to handle diverse camera projections (pinho
|
||||
sudo apt-get update && sudo apt-get install build-essential ffmpeg libsm6 libxext6 -y
|
||||
```
|
||||
|
||||
3. **Python Requirements**:
|
||||
3. **Python Requirements + optional dependencies** — one command sets up everything (base requirements, vggt, the SHARP submodule, gsplat, and the `inpainting_flux` sibling pack):
|
||||
|
||||
```bash
|
||||
pip install -r custom_nodes/camera-comfyUI/requirements.txt
|
||||
cd custom_nodes/camera-comfyUI && python install.py
|
||||
```
|
||||
|
||||
* *Optional:* `open3d` for GUI point cloud tools.
|
||||
This is the same script ComfyUI-Manager runs automatically; it is idempotent, and every optional step is non-fatal.
|
||||
|
||||
4. **Additional Nodes** (for certain workflows):
|
||||
**What it covers** (for reference — no manual action needed):
|
||||
|
||||
* Clone the following repositories directly into your `custom_nodes` folder:
|
||||
* [ComfyUI-Flux-Inpainting](https://github.com/rubi-du/ComfyUI-Flux-Inpainting)
|
||||
* [ComfyUI-Image-Filters](https://github.com/spacepxl/ComfyUI-Image-Filters)
|
||||
* **Important:** If the `ComfyUI-Flux-Inpainting` repository is cloned as `ComfyUI-Flux-Inpainting-main`, rename the folder to `inpainting_flux`:
|
||||
```bash
|
||||
mv custom_nodes/ComfyUI-Flux-Inpainting-main custom_nodes/inpainting_flux
|
||||
```
|
||||
* **gsplat** — CUDA-accelerated Gaussian splat rasterizer. Required by `SplatPolish`, SHARP, and the fast render backend for `RenderSplat` / `RenderSplats4D*`. Kernels JIT-compile on first use (needs a CUDA GPU + matching PyTorch build).
|
||||
* **vggt** — camera pose + depth estimation (`VideoPoseEstimator`). Not on PyPI — installed with `pip install git+https://github.com/facebookresearch/vggt.git`. A sibling clone of [facebookresearch/vggt](https://github.com/facebookresearch/vggt) in your ComfyUI root also works. The `facebook/VGGT-1B` weights (~5 GB) download via `huggingface_hub` on first use.
|
||||
* **CoTracker3** — point tracking for `EstimateTracks`. Fetched automatically via `torch.hub` on first use.
|
||||
* **SHARP** — image→splat prediction (`ImageToSplat`, `FisheyeToGaussian`, `VideoToFusedSplats`, `SplatTrajectoryEnricher`). Lives as the git submodule at `submodules/ml-sharpt` ([apple/ml-sharp](https://github.com/apple/ml-sharp)); `install.py` initializes it for you.
|
||||
* **ComfyUI-Flux-Inpainting** — cloned into `custom_nodes/inpainting_flux` if missing. Any of the usual folder names (`inpainting_flux`, `ComfyUI-Flux-Inpainting`, `ComfyUI-Flux-Inpainting-main`) is detected — no renaming needed.
|
||||
|
||||
5. **Flux Models** (Hugging Face):
|
||||
4. **Additional Nodes** (only for some example workflows):
|
||||
|
||||
* [ComfyUI-Image-Filters](https://github.com/spacepxl/ComfyUI-Image-Filters) — install via Manager or clone into `custom_nodes`.
|
||||
|
||||
5. **Flux Models** (Hugging Face, only for gated models):
|
||||
|
||||
```bash
|
||||
pip install huggingface_hub
|
||||
huggingface-cli login
|
||||
```
|
||||
|
||||
@@ -80,18 +101,45 @@ A collection of ComfyUI custom nodes to handle diverse camera projections (pinho
|
||||
* ### Reprojection Nodes
|
||||
|
||||
* `ReprojectImage`, `ReprojectDepth`, `OutpaintAnyProjection`
|
||||
|
||||
* ### Matrix Nodes
|
||||
|
||||
* `TransformToMatrix`, `TransformToMatrixManual`
|
||||
|
||||
* ### Depth Nodes
|
||||
|
||||
* `DepthEstimatorNode`, `DepthToImageNode`, `ZDepthToRayDepthNode`
|
||||
* `CombineDepthsNode`, `DepthRenormalizer`
|
||||
* `CombineDepthsNode`, `DepthRenormalizer`, `FisheyeDepthEstimator`
|
||||
|
||||
* ### Point Cloud Nodes
|
||||
|
||||
* `DepthToPointCloud`, `TransformPointCloud`, `ProjectPointCloud`
|
||||
* `PointCloudUnion`, `PointCloudCleaner`, `LoadPointCloud`, `SavePointCloud`
|
||||
* `DepthToPointCloud`, `TransformPointCloud`, `ProjectPointCloud`, `PointCloudUnion`
|
||||
* `PointCloudCleaner`, `LoadPointCloud`, `SavePointCloud`, `ProjectAndClean`, `DepthEdgeFilter`
|
||||
|
||||
* ### Trajectory Nodes
|
||||
|
||||
* `CameraMotionNode`, `CameraInterpolationNode`, `CameraTrajectoryNode`
|
||||
* `SaveTrajectory`, `LoadTrajectory`, `PointcloudTrajectoryEnricher`
|
||||
|
||||
* ### Gaussian Splat Nodes
|
||||
|
||||
* `LoadPlySplat`, `SavePlySplat`, `ImageToSplat`, `FisheyeToGaussian`
|
||||
* `RotateSplats`, `MergeSplats`, `FuseSplats`, `RenderSplat`
|
||||
* `VideoToFusedSplats`, `SplatPolish`
|
||||
|
||||
* ### 4D Gaussian Splat Nodes
|
||||
|
||||
* `MotionMaskFromDepth`, `EstimateTracks`, `TracksToTrajectories`, `SplitSplatsByMask`
|
||||
* `BuildSplats4D`, `RenderSplats4DFrame`, `RenderSplats4DVideo`
|
||||
* `SaveSplats4D`, `LoadSplats4D`
|
||||
|
||||
* ### Pose Nodes
|
||||
|
||||
* `VideoPoseEstimator`, `TrajectoryInvert`, `TrajectoryCompose`
|
||||
|
||||
* ### World Nodes
|
||||
|
||||
* `DepthScaleAnchor`, `SplatTrajectoryEnricher`, `SphereSplatSeed`
|
||||
|
||||
---
|
||||
|
||||
@@ -108,30 +156,97 @@ A collection of ComfyUI custom nodes to handle diverse camera projections (pinho
|
||||
| `DepthToPointCloud` | Converts Depth and image to → 3D point cloud tensor (N×7). |
|
||||
| `DepthToImageNode` | Converts depth to image (N×3) using a color map. |
|
||||
| `ZDepthToRayDepthNode` | Converts Z-depth (output of metric-depth-anything) to ray depth to compensate lens curvature. |
|
||||
| `TransformPointCloud` | Applies 4×4 rotation matrix to point cloud |
|
||||
| `TransformPointCloud` | Applies 4×4 rotation matrix to point cloud. |
|
||||
| `ProjectPointCloud` | Z-buffer–based projection of point cloud into image + mask. |
|
||||
| `CameraMotionNode` | Generates image sequences by moving camera along a trajectory. |
|
||||
| `PointCloudCleaner` | Removes isolated points via voxel filtering. |
|
||||
| `PointCloudUnion` | Combines multiple point clouds into one. |
|
||||
| `LoadPointCloud` | Loads a point cloud from `.npy` or `.ply` format. |
|
||||
| `SavePointCloud` | Saves a point cloud to `.npy` or `.ply` format. |
|
||||
| `CameraMotionNode` | Generates image and mask sequences along a camera trajectory with optional mask dilation/inversion. |
|
||||
| `CameraInterpolationNode` | Builds a trajectory tensor from two poses. |
|
||||
| `CameraTrajectoryNode` | Interactive Open3D GUI for recording camera waypoints. |
|
||||
| `PointCloudCleaner` | Removes isolated points via voxel filtering. |
|
||||
| `SaveTrajectory` | Saves a trajectory tensor to a file. |
|
||||
| `LoadTrajectory` | Loads a trajectory tensor from a file. |
|
||||
| `VideoCameraMotionSequence` | Processes video frames and depth maps along a camera trajectory, generating reprojected outputs. |
|
||||
| `DepthFramesToVideo` | Converts a sequence of depth maps into video frame tensors for saving. |
|
||||
| `VideoMetricDepthEstimate` | Estimates metric depth for a sequence of frames using VideoDepthAnything. |
|
||||
| `DepthEdgeFilter` | Detects "flying pixel" depth discontinuities and outputs a validity mask (1.0 = valid). |
|
||||
| `LoadPlySplat` | Loads a 3D Gaussian Splatting `.ply` file into a `GSPLAT` object. |
|
||||
| `SavePlySplat` | Saves a `GSPLAT` to the ComfyUI output directory as a `.ply` file. |
|
||||
| `ImageToSplat` | Predicts Gaussian splats from a single image using SHARP. |
|
||||
| `FisheyeToGaussian` | Reprojects a fisheye view to multiple pinhole angles, predicts splats, rotates and merges them. |
|
||||
| `RotateSplats` | Applies a 4×4 transform matrix to a splat cloud. |
|
||||
| `MergeSplats` | Concatenates two `GSPLAT` objects into one. |
|
||||
| `FuseSplats` | Fuses two splat clouds with weighted voxel merging (keep/discard/average/smart modes). |
|
||||
| `RenderSplat` | Renders a splat cloud from a camera pose into an image + mask. |
|
||||
| `VideoToFusedSplats` | Runs SHARP on video keyframes, scale-aligns to metric depth, filters dynamic pixels, and fuses all keyframes into one world-frame splat cloud. |
|
||||
| `SplatPolish` | Optimizes a world-frame splat cloud against posed video frames (L1 + D-SSIM) using gsplat's differentiable rasterizer. |
|
||||
| `MotionMaskFromDepth` | Detects dynamic pixels from a depth+pose sequence (1.0 = moving). |
|
||||
| `EstimateTracks` | Runs CoTracker3 on a video; returns tracks `[T,N,2]` (pixels) and visibility `[T,N]`. |
|
||||
| `TracksToTrajectories` | Unprojects 2D tracks with depth and camera poses into world-space 3D trajectories `[T,M,3]`. |
|
||||
| `SplitSplatsByMask` | Projects splat centers into a 2D mask and splits the cloud into inside/outside parts. |
|
||||
| `BuildSplats4D` | Builds a 4D splat scene: each canonical splat follows a kNN blend of track control-point motions. |
|
||||
| `RenderSplats4DFrame` | Evaluates the 4D scene at a single time value and renders it from a given camera. |
|
||||
| `RenderSplats4DVideo` | Interpolates the camera path, sweeps time from start to end, and renders each frame. |
|
||||
| `SaveSplats4D` | Saves a `GSPLAT4D` scene as an `.npz` archive (plus optional per-frame PLYs). |
|
||||
| `LoadSplats4D` | Loads a `GSPLAT4D` scene from an `.npz` archive. |
|
||||
| `VideoPoseEstimator` | VGGT-based per-frame camera poses `[T,4,4]`, depth maps, FOV and depth confidence from a video clip. |
|
||||
| `TrajectoryInvert` | Inverts each 4×4 pose (world-to-camera ↔ camera-to-world). |
|
||||
| `TrajectoryCompose` | Per-frame matrix product `A @ B`; a single 4×4 input broadcasts over the other. |
|
||||
| `DepthScaleAnchor` | Robustly aligns a depth map to a reference depth via disparity-domain scale(+shift). |
|
||||
| `SplatTrajectoryEnricher` | Expands a splat world along a trajectory: render, outpaint holes with Flux, lift with SHARP, scale-align, smart-stitch. |
|
||||
| `SphereSplatSeed` | Converts an equirectangular panorama into a Gaussian sphere seeding a 360° world. |
|
||||
|
||||
---
|
||||
|
||||
## Video → 4D World
|
||||
|
||||
Turn a monocular video into a navigable 4D (3D + time) Gaussian splat scene and re-render it from any novel camera trajectory. The reference workflow is **`workflows/video_to_4d_world.json`**; the stages are:
|
||||
|
||||
1. **Pose & depth (VGGT)** — `VideoPoseEstimator` estimates per-frame world-to-camera poses `[T,4,4]`, depth maps, FOV and depth confidence from the input frames. Since the depth maps are Z-depths, run `ZDepthToRayDepthNode` before any node that expects ray depth (see caveats below). `DepthEdgeFilter` can additionally mask out flying pixels at depth discontinuities.
|
||||
2. **Motion masking** — `MotionMaskFromDepth` warps depth between frames using the estimated poses and flags pixels whose residual is too large as dynamic (moving objects vs. static background).
|
||||
3. **Static splat fusion + polish** — `VideoToFusedSplats` runs SHARP on keyframes, keeps only static pixels (via the motion mask), scale-aligns each keyframe to metric depth, transforms splats into the world frame and fuses them incrementally. `SplatPolish` then fine-tunes the fused cloud photometrically against the posed video frames.
|
||||
4. **Tracked dynamic 4D Gaussians** — `EstimateTracks` (CoTracker3) tracks a dense point grid across the video; `TracksToTrajectories` lifts the tracks to world-space 3D using depth + poses; `SplitSplatsByMask` separates dynamic splats from the static background; `BuildSplats4D` binds the dynamic canonical splats to track control points via kNN blending, producing a `GSPLAT4D` scene.
|
||||
5. **Render a novel trajectory** — build any new camera path (e.g. `CameraInterpolationNode`, `TrajectoryCompose` to retarget relative to a source pose) and render with `RenderSplats4DVideo` (or single frames with `RenderSplats4DFrame`). Save/reload scenes with `SaveSplats4D` / `LoadSplats4D`.
|
||||
|
||||
**Static-camera fisheye variant** — for footage from a locked-off 180° fisheye camera, `workflows/fisheye_static_video_to_4d.json` skips pose estimation entirely (identity trajectory), uses the batched `FisheyeDepthEstimator` for per-frame radial depth and `FisheyeToGaussian` on frame 0 for the whole static world, then follows the same track → split → `BuildSplats4D` → render path (all 4D nodes accept the FISHEYE projection directly).
|
||||
|
||||
### Caveats
|
||||
|
||||
* **Z-depth vs ray depth**: depth estimators (including `VideoPoseEstimator`) output Z-depth; point-cloud and splat lifting nodes expect ray depth. Insert `ZDepthToRayDepthNode` where needed, or geometry will bow at wide FOVs.
|
||||
* **`SplatPolish` requires gsplat + CUDA**: without them it can fall back to the differentiable torch renderer at reduced resolution, which is extremely slow (minutes per 100 iterations).
|
||||
* **`EstimateTracks` downloads CoTracker3 via `torch.hub` on first use** — expect a one-time download and allow network access.
|
||||
* **`VideoPoseEstimator` downloads `facebook/VGGT-1B` (~5 GB)** on first use via `huggingface_hub`.
|
||||
|
||||
---
|
||||
|
||||
## Workflows
|
||||
|
||||
A set of JSON workflows illustrating typical use cases. Each workflow lives in `workflows/` and can be loaded directly in ComfyUI.
|
||||
A set of JSON workflows illustrating typical use cases. Once the pack is installed they appear in ComfyUI under **Workflow → Browse Templates** (with thumbnails); the files live in `workflows/` and can also be loaded directly. Each workflow contains an embedded **“About this workflow”** note in the canvas explaining its stages, what to set, and what it needs — and references the bundled example inputs that `install.py` copies into your ComfyUI `input/` folder, so they run as-is on a fresh install.
|
||||
|
||||
| Workflow | Description |
|
||||
| -------------------------------------- | -------------------------------------------------------------- |
|
||||
| **demo\_camera\_workflow\.json** | Masked reprojection demo: pinhole → fisheye/equirect |
|
||||
| **outpainting\_fisheye.json** | Text‐guided fisheye outpainting (built‐in inpaint node) |
|
||||
| **outpainting\_fisheye\_flux.json** | Flux‐based outpainting with clear reprojection scheme |
|
||||
| **Outpaint\_node\_test.json** | Test harness for the universal outpaint node |
|
||||
| **Outpaint\_fisheye180.json** | 180° fisheye outpainting via `OutpaintAnyProjection` |
|
||||
| **Fisheye\_depth\_workflow\.json** | Fisheye → metric depth → point cloud → PLY export |
|
||||
| **Pointcloud.json** | Metric‐depth‐anything v2 → point cloud → camera view synthesis |
|
||||
| **pointcloud\_inpaint.json** | Inpaint + backproject to 3D for dynamic camera motion videos |
|
||||
| **Pointcloud\_walker.json** | GUI‐based camera control via Open3D |
|
||||
*Extras* below means dependencies beyond this pack and Depth-Anything V2 (which auto-downloads); `inpainting_flux` is installed automatically by `install.py`.
|
||||
|
||||
| Workflow | Description | Extras |
|
||||
| --- | --- | --- |
|
||||
| **demo\_camera\_workflow\.json** | Minimal demo: rotate the camera and reproject pinhole → equirectangular, with coverage mask | — |
|
||||
| **Outpaint\_node\_test.json** | One-patch smoke test of `OutpaintAnyProjection` | inpainting_flux |
|
||||
| **Outpaint\_fisheye180.json** | Pinhole 90° → full 180° fisheye via five chained `OutpaintAnyProjection` passes + composite/upscale | inpainting_flux |
|
||||
| **outpainting\_fisheye\_flux.json** | Manual version of the above: explicit Flux Inpainting + reprojection stages | inpainting_flux, RealESRGAN |
|
||||
| **fisheye\_to\_pointcloud.json** | Fisheye 180° → metric depth → point cloud (`.ply`/`.npy`) | — |
|
||||
| **PointCloud.json** | Single image → point cloud → cleaned novel-view render | Image-Filters (optional) |
|
||||
| **pointcloud\_walker.json** | Image → point cloud → camera fly-through WEBM | — |
|
||||
| **test\_pointcloud\_loading.json** | Reload a saved point cloud and orbit-render it | — |
|
||||
| **record\_trajectory.json** | Record a camera trajectory `.npy` for LoadTrajectory (two poses → SE(3) interpolation → SaveTrajectory) | — |
|
||||
| **pointcloud\_inpaint.json** | Enrich a cloud: Flux-inpaint disocclusions, lift them to 3D, merge, orbit render | inpainting_flux |
|
||||
| **PC\_enricher.json** | One-node version of the above: `PointcloudTrajectoryEnricher` along a saved trajectory | inpainting_flux |
|
||||
| **sbs180\_workflow.json** | Synthesize the second eye of a VR180 stereo pair from one fisheye view | inpainting_flux |
|
||||
| **video\_camera.json** | Re-shoot a video with a new camera move; WAN VACE regenerates disocclusions, Florence2 auto-captions | VHS, Florence2; WAN 2.1 VACE + Video-Depth-Anything models |
|
||||
| **wan\_vace\_ref\_to\_video.json** | Still fisheye image + recorded trajectory → WAN VACE camera-move video | WAN 2.1 VACE models; VHS (optional MP4 export) |
|
||||
| **video_to_4d_world\.json** | Video → 4D world: VGGT poses/depth → motion masking → fused static splats + polish → tracked dynamic 4D Gaussians → novel-trajectory render. | — |
|
||||
| **video_to_4d_walkable_world\.json** | Video → 4D WALKABLE world (test-friendly defaults): polished static splats enriched along a walk trajectory (`SplatTrajectoryEnricher`, Flux outpaint + SHARP) → 4D scene → walk-through render + `.ply`/`.npz` exports for free walking in external 3DGS viewers. | inpainting_flux |
|
||||
| **fisheye\_static\_video\_to\_4d.json** | Static-camera 180° fisheye video → 4D Gaussian scene → novel-path render. No pose estimation needed: identity trajectory, batched fisheye depth, frame-0 splats split into static world + dynamic canonical. | VHS |
|
||||
|
||||
Superseded reference graphs live in `workflows/legacy/` (kept out of the template browser): **outpainting\_fisheye.json** (SD-inpaint-checkpoint variant of the flux outpaint) and **Fisheye\_depth\_workflow\.json** (the multi-view depth fusion that `FisheyeDepthEstimator` now performs internally).
|
||||
|
||||
---
|
||||
|
||||
@@ -146,9 +261,9 @@ Basic reprojection pipeline: apply masks, rotate pinhole camera, outpaint fishey
|
||||
<img src="demo_images/Pinhole_camera_rotation.png" alt="Pinhole Rotation" width="45%" />
|
||||
</div>
|
||||
|
||||
### 2. `outpainting_fisheye.json`
|
||||
### 2. `legacy/outpainting_fisheye.json`
|
||||
|
||||
Simplest text‐guided fisheye outpainting built with the core inpaint node.
|
||||
Simplest text‐guided fisheye outpainting built with the core inpaint node (superseded by the Flux variant).
|
||||
|
||||
### 3. `outpainting_fisheye_flux.json`
|
||||
|
||||
@@ -164,9 +279,9 @@ Flux Inpainting ensures sharper results and explicit reprojection stages.
|
||||
|
||||
<img src="demo_images/Fisheye_outpainted_flux_dev.png" alt="Flux Dev" width="60%" />
|
||||
|
||||
### 5. `Fisheye_depth_workflow.json`
|
||||
### 5. `legacy/Fisheye_depth_workflow.json`
|
||||
|
||||
Convert fisheye images to metric depth and generate a PLY point cloud.
|
||||
Convert fisheye images to metric depth and generate a PLY point cloud — the manual multi-view graph that `FisheyeDepthEstimator` now performs in one node (see `fisheye_to_pointcloud.json`).
|
||||
|
||||
<img src="demo_images/Depthmap.png" alt="Fisheye Depth→PointCloud" width="60%" />
|
||||
|
||||
@@ -176,7 +291,7 @@ Convert fisheye images to metric depth and generate a PLY point cloud.
|
||||
|
||||
Quick test for the universal outpaint node in arbitrary views and camera movement
|
||||
|
||||
### 7. `Pointcloud.json`
|
||||
### 7. `PointCloud.json`
|
||||
|
||||
Depth→PointCloud pipeline with interactive camera movement and reprojection views.
|
||||
|
||||
@@ -190,9 +305,54 @@ Inpaint image with shifted camera and backproject for dynamic camera‐driven vi
|
||||
<img src="demo_images/Fisheye_camera_pointcloud_moved_outpainted.png" alt="PointCloud Inpaint" width="40%" />
|
||||
<img src="demo_images/Camera_interpolation_pointcloud.gif" alt="PointCloud Inpaint Video" width="40%" />
|
||||
|
||||
### 10. `Pointcloud_walker.json`
|
||||
### 9. `sbs180_workflow.json`
|
||||
|
||||
Interactive Open3D-based GUI for walking and setting camera trajectory inside pointcloud.
|
||||
Take a wide-angle (fisheye or equirectangular) high-resolution (e.g., 4096×4096) image and generate a stereo pair by moving the camera horizontally. The output is a wide-angle stereo pair (side-by-side), simulating a fisheye or equirectangular stereo camera.
|
||||
|
||||
<img src="demo_images/equirect_stereo.gif" alt="Equirectangular Stereo Demo" width="80%" />
|
||||
|
||||
### 10. `pointcloud_walker.json`
|
||||
|
||||
Image → point cloud → camera fly-through rendered to WEBM (`CameraTrajectoryNode` + `CameraMotionNode`).
|
||||
|
||||
### 11. `video_camera.json`
|
||||
|
||||
This workflow demonstrates camera trajectory movement using the `wan-vace` video inpainting model. It generates smooth camera movements along a trajectory while filling missing regions with high-quality inpainting.
|
||||
|
||||
<div style="display:flex; gap:10px;">
|
||||
<img src="demo_images/camera_movement.gif" alt="Camera Movement Demo" width="80%" />
|
||||
</div>
|
||||
|
||||
---
|
||||
|
||||
## Trajectory Concept
|
||||
|
||||
A **trajectory** in camera-comfyUI is a sequence of camera poses, each represented as a 4×4 transformation matrix. This set of matrices defines the path and orientation of the camera through 3D space, enabling smooth and complex camera movements for view synthesis, point cloud rendering, and video generation.
|
||||
|
||||
### Creating Trajectories
|
||||
|
||||
There are two main ways to create a trajectory:
|
||||
|
||||
- **Camera Matrices Interpolation:**
|
||||
Define two or more camera poses (as matrices), and interpolate between them to generate a smooth path. The `CameraInterpolationNode` automates this process, producing a trajectory tensor for use in camera motion nodes.
|
||||
|
||||
- **Walking in Open3D Environment:**
|
||||
Use the interactive Open3D GUI (`CameraTrajectoryNode`) to "walk" through the point cloud. As you move the camera, waypoints (poses) are recorded, forming a trajectory that can be exported and reused.
|
||||
|
||||
### Using Trajectories
|
||||
|
||||
The `CameraMotionNode` takes a trajectory (set of matrices) and interpolates camera positions and orientations along it, producing smooth camera movements for rendering sequences or videos.
|
||||
|
||||
---
|
||||
|
||||
## Point Cloud Formats
|
||||
|
||||
Point clouds can be saved and loaded in two formats:
|
||||
|
||||
- **.npy**: Numpy array format (fast, preserves all tensor data, recommended for internal pipelines).
|
||||
- **.ply**: Polygon File Format (widely supported, viewable in external 3D tools).
|
||||
|
||||
Use the `SavePointCloud` and `LoadPointCloud` nodes to handle I/O operations in either format.
|
||||
|
||||
---
|
||||
|
||||
@@ -202,11 +362,15 @@ Contributions welcome! Please open issues or PRs to add features, improve docs,
|
||||
|
||||
## TODO List
|
||||
|
||||
* [ ] Add processing to pointcloud or depthmap to remove outlier and lonely points at depth borders.
|
||||
* [x] Add processing to pointcloud or depthmap to remove outlier and lonely points at depth borders.
|
||||
* [x] Use built-in comfyUI mask type an image.
|
||||
* [x] Unite nodes into groups to simplify workflows.
|
||||
* [ ] Create a single workflow for view synthesis.
|
||||
* [x] Create a single workflow for view synthesis (`video_to_4d_world.json`).
|
||||
* [x] Implement easier and more flexible camera control - more complex camera movements with more than 2 points.
|
||||
* [x] Add more examples and documentation for each node.
|
||||
* [x] Add pointcloud union
|
||||
* [ ] Fix imports for renamed folders (e.g., inpainting_flux)
|
||||
* [x] Fix imports for renamed folders (e.g., inpainting_flux)
|
||||
* [x] Integrate camera movement pipeline with video models (e.g., wan2.1) for smooth, high-quality inpainting along camera trajectories.
|
||||
* [ ] Compressed export format for 4D scenes (current `.npz` stores raw tensors).
|
||||
* [ ] SAM2-based refinement of motion masks (current masks come from depth-warp residuals only).
|
||||
* [ ] Fisheye/equirectangular rendering through gsplat (e.g., via cubemap render + reprojection); the fast CUDA path is currently pinhole-only.
|
||||
|
||||
@@ -3,6 +3,28 @@ from .reprojection_nodes import NODE_CLASS_MAPPINGS as NCM2
|
||||
from .metric_depth_nodes import NODE_CLASS_MAPPINGS as NCM3
|
||||
from .flux_fisheye_filling_nodes import NODE_CLASS_MAPPINGS as NCM4
|
||||
from .complex_nodes import NODE_CLASS_MAPPINGS as NCM5
|
||||
NODE_CLASS_MAPPINGS = {**NCM1, **NCM2, **NCM3, **NCM4, **NCM5}
|
||||
from .video_nodes import NODE_CLASS_MAPPINGS as NCM6
|
||||
from .GS_nodes import NODE_CLASS_MAPPINGS as NCM7
|
||||
|
||||
__all__ = ["NODE_CLASS_MAPPINGS"]
|
||||
# Optional node packs: a missing/broken optional dependency must never kill the
|
||||
# whole extension (mirrors how video_nodes degrades when video_depth_anything
|
||||
# is unavailable).
|
||||
try:
|
||||
from .GS4D_nodes import NODE_CLASS_MAPPINGS as NCM8
|
||||
except Exception as _exc:
|
||||
print(f"[camera-comfyUI] Warning: GS4D_nodes could not be loaded, 4D splat nodes disabled: {_exc}")
|
||||
NCM8 = {}
|
||||
try:
|
||||
from .pose_nodes import NODE_CLASS_MAPPINGS as NCM9
|
||||
except Exception as _exc:
|
||||
print(f"[camera-comfyUI] Warning: pose_nodes could not be loaded, pose estimation nodes disabled: {_exc}")
|
||||
NCM9 = {}
|
||||
try:
|
||||
from .world_nodes import NODE_CLASS_MAPPINGS as NCM10
|
||||
except Exception as _exc:
|
||||
print(f"[camera-comfyUI] Warning: world_nodes could not be loaded, world-building nodes disabled: {_exc}")
|
||||
NCM10 = {}
|
||||
|
||||
NODE_CLASS_MAPPINGS = {**NCM1, **NCM2, **NCM3, **NCM4, **NCM5, **NCM6, **NCM7, **NCM8, **NCM9, **NCM10}
|
||||
|
||||
__all__ = ["NODE_CLASS_MAPPINGS"]
|
||||
|
||||
@@ -64,7 +64,7 @@ class FisheyeDepthEstimator:
|
||||
RETURN_TYPES = ("TENSOR","MASK")
|
||||
RETURN_NAMES = ("depthmap","mask")
|
||||
FUNCTION = "estimate_fisheye_depth"
|
||||
CATEGORY = "Camera/depth"
|
||||
CATEGORY = "Camera/Depth"
|
||||
|
||||
def estimate_fisheye_depth(
|
||||
self,
|
||||
@@ -95,46 +95,76 @@ class FisheyeDepthEstimator:
|
||||
depth_full, = de_node.estimate_depth(image, model_name, depth_scale)
|
||||
mask_full = (depth_full > 0).float()
|
||||
|
||||
# 2) Generate Pinhole Views
|
||||
fisheye_depths, fisheye_masks = self._generate_pinhole_views(
|
||||
image,
|
||||
de_node, z2r_node, ri_node, rd_node,
|
||||
fisheye_fov, pinhole_fov,
|
||||
pin_w, pin_h, fish_w, fish_h,
|
||||
model_name, depth_scale, median_blur_kernel
|
||||
)
|
||||
# 2) Pinhole orientations (5 views)
|
||||
rotations = [
|
||||
(0, 0, 0), # front
|
||||
(0, 45, 0), # right
|
||||
(0, -45, 0), # left
|
||||
(45, 0, 0), # up
|
||||
(-45, 0, 0), # down
|
||||
]
|
||||
|
||||
fisheye_depths = []
|
||||
fisheye_masks = []
|
||||
|
||||
# euler → matrix
|
||||
def euler_to_matrix(pitch, yaw, roll):
|
||||
p, y, r = map(math.radians, (pitch, yaw, roll))
|
||||
Rx = torch.tensor([[1,0,0],[0,math.cos(p),-math.sin(p)],[0,math.sin(p),math.cos(p)]], dtype=torch.float32)
|
||||
Ry = torch.tensor([[math.cos(y),0,math.sin(y)],[0,1,0],[-math.sin(y),0,math.cos(y)]], dtype=torch.float32)
|
||||
Rz = torch.tensor([[math.cos(r),-math.sin(r),0],[math.sin(r),math.cos(r),0],[0,0,1]], dtype=torch.float32)
|
||||
R = Rz @ Ry @ Rx
|
||||
M = torch.eye(4, dtype=torch.float32)
|
||||
M[:3, :3] = R
|
||||
return M
|
||||
|
||||
# 3) Process each orientation
|
||||
for pitch, yaw, roll in rotations:
|
||||
M = euler_to_matrix(pitch, yaw, roll)
|
||||
M_np = M.numpy()
|
||||
M_inv = torch.inverse(M).numpy()
|
||||
|
||||
# fisheye → pinhole
|
||||
img_pin, mask_pin = ri_node.reproject_image(
|
||||
image,
|
||||
input_horiszontal_fov = fisheye_fov,
|
||||
output_horiszontal_fov= pinhole_fov,
|
||||
input_projection = "FISHEYE",
|
||||
output_projection = "PINHOLE",
|
||||
output_width = pin_w,
|
||||
output_height = pin_h,
|
||||
transform_matrix = M_np,
|
||||
feathering = 0,
|
||||
)
|
||||
|
||||
# estimate pinhole depth
|
||||
depth_pin, = de_node.estimate_depth(img_pin, model_name, depth_scale, median_blur_kernel=median_blur_kernel)
|
||||
depth_pin, = z2r_node.depth_to_ray_depth(
|
||||
depth_pin,
|
||||
pinhole_fov,
|
||||
)
|
||||
# pinhole → fisheye
|
||||
fish_depth, fish_mask = rd_node.reproject_depth(
|
||||
depth_pin,
|
||||
input_horizontal_fov = pinhole_fov,
|
||||
output_horizontal_fov= fisheye_fov,
|
||||
input_projection = "PINHOLE",
|
||||
output_projection = "FISHEYE",
|
||||
output_width = fish_w,
|
||||
output_height = fish_h,
|
||||
transform_matrix = M_inv,
|
||||
)
|
||||
# squeeze mask to [B,H,W]
|
||||
fish_mask = fish_mask.squeeze(1)
|
||||
|
||||
fisheye_depths.append(fish_depth) # [B,H,W]
|
||||
fisheye_masks.append(fish_mask)
|
||||
fisheye_depths.append(depth_full) # [B,H,W]
|
||||
fisheye_masks.append(mask_full.squeeze(-1)) # [B,H,W 1]
|
||||
# merged mask
|
||||
merged_mask = torch.sum(torch.stack(fisheye_masks), dim=0) > 0.5
|
||||
# print(fisheye_depths[0].shape, fisheye_depths[-1].shape, merged_mask.shape)
|
||||
# 4) Merge in sequence
|
||||
d_acc, m_acc = self._merge_depths(
|
||||
fisheye_depths, fisheye_masks,
|
||||
ren_node, comb_node,
|
||||
mode, softmerge_radius
|
||||
)
|
||||
|
||||
# 5) Circular mask
|
||||
ys = torch.arange(fish_h, device=d_acc.device).view(1, fish_h, 1)
|
||||
xs = torch.arange(fish_w, device=d_acc.device).view(1, 1, fish_w)
|
||||
cy = (fish_h - 1) / 2.0
|
||||
cx = (fish_w - 1) / 2.0
|
||||
dist2 = (ys - cy)**2 + (xs - cx)**2
|
||||
radius2 = (min(fish_w, fish_h) / 2.0)**2
|
||||
circ_mask = (dist2 <= radius2).float()
|
||||
return d_acc, circ_mask
|
||||
|
||||
def _merge_depths(
|
||||
self,
|
||||
fisheye_depths: list,
|
||||
fisheye_masks: list,
|
||||
ren_node: DepthRenormalizer,
|
||||
comb_node: CombineDepthsNode,
|
||||
mode: str,
|
||||
softmerge_radius: int,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
d_acc = fisheye_depths[0]
|
||||
m_acc = fisheye_masks[0]
|
||||
for d_new, m_new in zip(fisheye_depths[1:-1], fisheye_masks[1:-1]):
|
||||
@@ -157,120 +187,20 @@ class FisheyeDepthEstimator:
|
||||
m_acc,
|
||||
d_norm,
|
||||
m_new_last,
|
||||
mode = "SRC", # Use SRC for the full fisheye to preserve its details
|
||||
mode = "SRC",
|
||||
invert_mask = False,
|
||||
softmerge_radius = softmerge_radius
|
||||
)
|
||||
return d_acc, m_acc
|
||||
|
||||
def _generate_pinhole_views(
|
||||
self,
|
||||
image: torch.Tensor,
|
||||
de_node: DepthEstimatorNode,
|
||||
z2r_node: ZDepthToRayDepthNode,
|
||||
ri_node: ReprojectImage,
|
||||
rd_node: ReprojectDepth,
|
||||
fisheye_fov: float,
|
||||
pinhole_fov: float,
|
||||
pin_w: int,
|
||||
pin_h: int,
|
||||
fish_w: int,
|
||||
fish_h: int,
|
||||
model_name: str,
|
||||
depth_scale: float,
|
||||
median_blur_kernel: int,
|
||||
) -> Tuple[list, list]:
|
||||
rotations = [
|
||||
(0, 0, 0), # front
|
||||
(0, 45, 0), # right
|
||||
(0, -45, 0), # left
|
||||
(45, 0, 0), # up
|
||||
(-45, 0, 0), # down
|
||||
]
|
||||
|
||||
fisheye_depths = []
|
||||
fisheye_masks = []
|
||||
|
||||
for pitch, yaw, roll in rotations:
|
||||
M = self._euler_to_matrix(pitch, yaw, roll)
|
||||
M_np = M.numpy()
|
||||
M_inv = torch.inverse(M).numpy()
|
||||
|
||||
fish_depth, fish_mask = self._process_view(
|
||||
image, M_np, M_inv,
|
||||
de_node, z2r_node, ri_node, rd_node,
|
||||
fisheye_fov, pinhole_fov,
|
||||
pin_w, pin_h, fish_w, fish_h,
|
||||
model_name, depth_scale, median_blur_kernel
|
||||
)
|
||||
|
||||
fisheye_depths.append(fish_depth)
|
||||
fisheye_masks.append(fish_mask)
|
||||
|
||||
return fisheye_depths, fisheye_masks
|
||||
|
||||
def _process_view(
|
||||
self,
|
||||
image: torch.Tensor,
|
||||
M_np: np.ndarray,
|
||||
M_inv: np.ndarray,
|
||||
de_node: DepthEstimatorNode,
|
||||
z2r_node: ZDepthToRayDepthNode,
|
||||
ri_node: ReprojectImage,
|
||||
rd_node: ReprojectDepth,
|
||||
fisheye_fov: float,
|
||||
pinhole_fov: float,
|
||||
pin_w: int,
|
||||
pin_h: int,
|
||||
fish_w: int,
|
||||
fish_h: int,
|
||||
model_name: str,
|
||||
depth_scale: float,
|
||||
median_blur_kernel: int,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
# fisheye → pinhole
|
||||
img_pin, mask_pin = ri_node.reproject_image(
|
||||
image,
|
||||
input_horiszontal_fov = fisheye_fov,
|
||||
output_horiszontal_fov= pinhole_fov,
|
||||
input_projection = "FISHEYE",
|
||||
output_projection = "PINHOLE",
|
||||
output_width = pin_w,
|
||||
output_height = pin_h,
|
||||
transform_matrix = M_np,
|
||||
feathering = 0,
|
||||
)
|
||||
|
||||
# estimate pinhole depth
|
||||
depth_pin, = de_node.estimate_depth(img_pin, model_name, depth_scale, median_blur_kernel=median_blur_kernel)
|
||||
depth_pin, = z2r_node.depth_to_ray_depth(
|
||||
depth_pin,
|
||||
pinhole_fov,
|
||||
)
|
||||
# pinhole → fisheye
|
||||
fish_depth, fish_mask = rd_node.reproject_depth(
|
||||
depth_pin,
|
||||
input_horizontal_fov = pinhole_fov,
|
||||
output_horizontal_fov= fisheye_fov,
|
||||
input_projection = "PINHOLE",
|
||||
output_projection = "FISHEYE",
|
||||
output_width = fish_w,
|
||||
output_height = fish_h,
|
||||
transform_matrix = M_inv,
|
||||
)
|
||||
# squeeze mask to [B,H,W]
|
||||
fish_mask = fish_mask.squeeze(1)
|
||||
return fish_depth, fish_mask
|
||||
|
||||
def _euler_to_matrix(self, pitch, yaw, roll):
|
||||
p, y, r = map(math.radians, (pitch, yaw, roll))
|
||||
Rx = torch.tensor([[1,0,0],[0,math.cos(p),-math.sin(p)],[0,math.sin(p),math.cos(p)]], dtype=torch.float32)
|
||||
Ry = torch.tensor([[math.cos(y),0,math.sin(y)],[0,1,0],[-math.sin(y),0,math.cos(y)]], dtype=torch.float32)
|
||||
Rz = torch.tensor([[math.cos(r),-math.sin(r),0],[math.sin(r),math.cos(r),0],[0,0,1]], dtype=torch.float32)
|
||||
R = Rz @ Ry @ Rx
|
||||
M = torch.eye(4, dtype=torch.float32)
|
||||
M[:3, :3] = R
|
||||
return M
|
||||
# 5) Circular mask
|
||||
ys = torch.arange(fish_h, device=d_acc.device).view(1, fish_h, 1)
|
||||
xs = torch.arange(fish_w, device=d_acc.device).view(1, 1, fish_w)
|
||||
cy = (fish_h - 1) / 2.0
|
||||
cx = (fish_w - 1) / 2.0
|
||||
dist2 = (ys - cy)**2 + (xs - cx)**2
|
||||
radius2 = (min(fish_w, fish_h) / 2.0)**2
|
||||
circ_mask = (dist2 <= radius2).float()
|
||||
return d_acc, circ_mask
|
||||
|
||||
class PointcloudTrajectoryEnricher:
|
||||
"""
|
||||
@@ -311,7 +241,7 @@ class PointcloudTrajectoryEnricher:
|
||||
RETURN_TYPES = ("TENSOR","IMAGE","TENSOR")
|
||||
RETURN_NAMES = ("enriched_pointcloud","debug_image","debug_depth")
|
||||
FUNCTION = "enrich_trajectory"
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/Trajectory"
|
||||
|
||||
def enrich_trajectory(
|
||||
self,
|
||||
@@ -358,244 +288,106 @@ class PointcloudTrajectoryEnricher:
|
||||
debug_img = torch.zeros((1, height, width, 3), device=device)
|
||||
debug_depth = torch.zeros((1, height, width, 1), device=device)
|
||||
enriched_pc = pointcloud
|
||||
# Initialize debug_img and debug_depth which will be updated in the loop
|
||||
# and will hold the values from the last processed view.
|
||||
debug_img = torch.zeros((1, height, width, 3), device=device)
|
||||
debug_depth = torch.zeros((1, height, width, 1), device=device)
|
||||
|
||||
# loop over trajectory (limit or full)
|
||||
for M in tqdm(trajectory[:15], desc="Enriching trajectory"):
|
||||
enriched_pc, view_debug_img, view_debug_depth = self._process_single_view(
|
||||
M, enriched_pc, device,
|
||||
proj_node, outpaint_node, depth_node, renorm_node,
|
||||
depth2pc_node, transform_node, clean_node, zdepth_node,
|
||||
camera_type, horizontal_fov, width, height,
|
||||
patch_projection, patch_horiz_fov, patch_res,
|
||||
patch_phi, patch_theta, prompt,
|
||||
num_inference_steps, guidance_scale, mask_blur,
|
||||
voxel_size, min_points_per_voxel, model_name
|
||||
M_np = M.cpu().numpy()
|
||||
M_inv = np.linalg.inv(M_np)
|
||||
|
||||
# transform and select front points
|
||||
rotated, = transform_node.transform_pointcloud(enriched_pc, M_np)
|
||||
pc_front = rotated[rotated[:, 2] > 0]
|
||||
|
||||
# clean front points
|
||||
pc_front, = clean_node.clean_pointcloud(
|
||||
pc_front,
|
||||
voxel_size=voxel_size,
|
||||
min_points_per_voxel=min_points_per_voxel,
|
||||
width=4096,
|
||||
height=4096,
|
||||
)
|
||||
debug_img = view_debug_img
|
||||
debug_depth = view_debug_depth
|
||||
return enriched_pc, debug_img, debug_depth
|
||||
|
||||
def _process_single_view(
|
||||
self,
|
||||
M_matrix: torch.Tensor,
|
||||
current_enriched_pc: torch.Tensor,
|
||||
device: torch.device,
|
||||
proj_node: ProjectPointCloud,
|
||||
outpaint_node: OutpaintAnyProjection,
|
||||
depth_node: DepthEstimatorNode,
|
||||
renorm_node: DepthRenormalizer,
|
||||
depth2pc_node: DepthToPointCloud,
|
||||
transform_node: TransformPointCloud,
|
||||
clean_node: PointCloudCleaner,
|
||||
zdepth_node: ZDepthToRayDepthNode,
|
||||
camera_type: str,
|
||||
horizontal_fov: float,
|
||||
width: int,
|
||||
height: int,
|
||||
patch_projection: str,
|
||||
patch_horiz_fov: float,
|
||||
patch_res: int,
|
||||
patch_phi: float,
|
||||
patch_theta: float,
|
||||
prompt: str,
|
||||
num_inference_steps: int,
|
||||
guidance_scale: float,
|
||||
mask_blur: int,
|
||||
voxel_size: float,
|
||||
min_points_per_voxel: int,
|
||||
model_name: str,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
M_np = M_matrix.cpu().numpy()
|
||||
M_inv = np.linalg.inv(M_np)
|
||||
# project to image + depth
|
||||
img, mask, depth_map = proj_node.project_pointcloud(
|
||||
pc_front,
|
||||
camera_type,
|
||||
horizontal_fov,
|
||||
width,
|
||||
height,
|
||||
point_size=3,
|
||||
return_inverse_depth=False,
|
||||
)
|
||||
debug_img = img
|
||||
# fill nan in depthmap with (-1)
|
||||
# outpaint missing regions
|
||||
hole_mask = (mask < 0.5).float()
|
||||
out_img, out_mask = outpaint_node.outpaint_any(
|
||||
img,
|
||||
input_projection = camera_type,
|
||||
input_horiz_fov = horizontal_fov,
|
||||
output_projection = camera_type,
|
||||
output_horiz_fov = horizontal_fov,
|
||||
output_width = width,
|
||||
output_height = height,
|
||||
patch_projection = patch_projection,
|
||||
patch_horiz_fov = patch_horiz_fov,
|
||||
patch_res = patch_res,
|
||||
patch_phi = patch_phi,
|
||||
patch_theta = patch_theta,
|
||||
prompt = prompt,
|
||||
num_inference_steps = num_inference_steps,
|
||||
cached = False,
|
||||
guidance_scale = guidance_scale,
|
||||
mask_blur = mask_blur,
|
||||
mask = hole_mask,
|
||||
debug = False,
|
||||
)
|
||||
debug_img = out_img
|
||||
# estimate and renormalize depth
|
||||
nan_mask = torch.isnan(depth_map)
|
||||
# …and replace them with –1.0 (in-place)
|
||||
depth_map[nan_mask] = 0
|
||||
# clip from -1 to 1000
|
||||
depth_map = torch.clamp(depth_map, 0, 1000.0)
|
||||
new_depth, = depth_node.estimate_depth(out_img, model_name, depth_scale=1.0)
|
||||
new_depth, = zdepth_node.depth_to_ray_depth(
|
||||
new_depth,
|
||||
horizontal_fov,
|
||||
)
|
||||
# renormalize depth
|
||||
norm_depth, = renorm_node.renormalize_depth(
|
||||
new_depth,
|
||||
depth_map,
|
||||
depth_mask=(mask>=0.5)*1,
|
||||
guidance_mask=(mask<0.5)*1,
|
||||
use_inverse=False,
|
||||
)
|
||||
# median blur on depth
|
||||
k = 5
|
||||
d = norm_depth.permute(0,3,1,2) # [B,1,H,W]
|
||||
pad = k//2
|
||||
pd = F.pad(d, (pad, pad, pad, pad), mode='reflect')
|
||||
patches = pd.unfold(2, k, 1).unfold(3, k, 1)
|
||||
patches = patches.contiguous().view(d.shape[0], d.shape[1], d.shape[2], d.shape[3], k*k)
|
||||
d, _ = patches.median(dim=-1)
|
||||
norm_depth = d.permute(0,2,3,1) # [B,H,W,1]
|
||||
debug_depth = norm_depth*hole_mask.unsqueeze(0).unsqueeze(-1)+depth_map*(1-hole_mask.unsqueeze(0).unsqueeze(-1))
|
||||
|
||||
img, mask, depth_map, pc_front = self._prepare_view_data(
|
||||
current_enriched_pc, M_np, transform_node, clean_node, proj_node,
|
||||
voxel_size, min_points_per_voxel, camera_type, horizontal_fov,
|
||||
width, height
|
||||
)
|
||||
# back to pointcloud
|
||||
pc_new, = depth2pc_node.depth_to_pointcloud(
|
||||
out_img,
|
||||
camera_type,
|
||||
horizontal_fov,
|
||||
depth_scale=1.0,
|
||||
invert_depth=False,
|
||||
depthmap=norm_depth,
|
||||
mask=hole_mask,
|
||||
)
|
||||
|
||||
# fill nan in depthmap with (-1)
|
||||
# outpaint missing regions
|
||||
hole_mask = (mask < 0.5).float()
|
||||
out_img = self._outpaint_missing_regions(
|
||||
img, hole_mask, # Pass hole_mask instead of the full mask
|
||||
outpaint_node, camera_type, horizontal_fov, width, height,
|
||||
patch_projection, patch_horiz_fov, patch_res,
|
||||
patch_phi, patch_theta, prompt,
|
||||
num_inference_steps, guidance_scale, mask_blur
|
||||
)
|
||||
# estimate and renormalize depth
|
||||
norm_depth, debug_depth_view = self._estimate_and_refine_depth(
|
||||
out_img, depth_map, mask, hole_mask,
|
||||
depth_node, zdepth_node, renorm_node,
|
||||
model_name, horizontal_fov
|
||||
)
|
||||
|
||||
# back to pointcloud
|
||||
pc_world = self._convert_depth_to_world_pointcloud(
|
||||
out_img, norm_depth, hole_mask, M_inv,
|
||||
depth2pc_node, transform_node,
|
||||
camera_type, horizontal_fov
|
||||
)
|
||||
# enriched_pc is not rotated
|
||||
current_enriched_pc = torch.cat([current_enriched_pc, pc_world.to(device)], dim=0)
|
||||
return current_enriched_pc, out_img, debug_depth_view # Return out_img and the depth for this view
|
||||
|
||||
def _convert_depth_to_world_pointcloud(
|
||||
self,
|
||||
out_img: torch.Tensor,
|
||||
norm_depth: torch.Tensor,
|
||||
hole_mask: torch.Tensor,
|
||||
M_inv: np.ndarray,
|
||||
depth2pc_node: DepthToPointCloud,
|
||||
transform_node: TransformPointCloud,
|
||||
camera_type: str,
|
||||
horizontal_fov: float,
|
||||
) -> torch.Tensor:
|
||||
pc_new, = depth2pc_node.depth_to_pointcloud(
|
||||
out_img,
|
||||
camera_type,
|
||||
horizontal_fov,
|
||||
depth_scale=1.0,
|
||||
invert_depth=False,
|
||||
depthmap=norm_depth,
|
||||
mask=hole_mask,
|
||||
)
|
||||
pc_world, = transform_node.transform_pointcloud(pc_new, M_inv)
|
||||
return pc_world
|
||||
|
||||
def _estimate_and_refine_depth(
|
||||
self,
|
||||
out_img: torch.Tensor,
|
||||
depth_map: torch.Tensor,
|
||||
original_mask: torch.Tensor, # Mask from projection
|
||||
hole_mask: torch.Tensor,
|
||||
depth_node: DepthEstimatorNode,
|
||||
zdepth_node: ZDepthToRayDepthNode,
|
||||
renorm_node: DepthRenormalizer,
|
||||
model_name: str,
|
||||
horizontal_fov: float,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
nan_mask = torch.isnan(depth_map)
|
||||
depth_map[nan_mask] = 0 # In-place modification
|
||||
depth_map = torch.clamp(depth_map, 0, 1000.0)
|
||||
|
||||
new_depth, = depth_node.estimate_depth(out_img, model_name, depth_scale=1.0)
|
||||
new_depth, = zdepth_node.depth_to_ray_depth(
|
||||
new_depth,
|
||||
horizontal_fov,
|
||||
)
|
||||
# renormalize depth
|
||||
# Use original_mask for depth_mask as it represents valid projected areas
|
||||
norm_depth, = renorm_node.renormalize_depth(
|
||||
new_depth,
|
||||
depth_map,
|
||||
depth_mask=(original_mask >= 0.5) * 1,
|
||||
guidance_mask=(hole_mask >= 0.5) * 1, # hole_mask is appropriate here
|
||||
use_inverse=False,
|
||||
)
|
||||
# median blur on depth
|
||||
k = 5
|
||||
d = norm_depth.permute(0,3,1,2) # [B,1,H,W]
|
||||
pad = k//2
|
||||
pd = F.pad(d, (pad, pad, pad, pad), mode='reflect')
|
||||
patches = pd.unfold(2, k, 1).unfold(3, k, 1)
|
||||
patches = patches.contiguous().view(d.shape[0], d.shape[1], d.shape[2], d.shape[3], k*k)
|
||||
d_median, _ = patches.median(dim=-1) # Renamed to avoid conflict
|
||||
norm_depth_blurred = d_median.permute(0,2,3,1) # [B,H,W,1]
|
||||
|
||||
# Create debug_depth_view using the blurred normalized depth for holes
|
||||
# and the original depth_map for non-holes.
|
||||
debug_depth_view = norm_depth_blurred * hole_mask.unsqueeze(0).unsqueeze(-1) + \
|
||||
depth_map * (1 - hole_mask.unsqueeze(0).unsqueeze(-1))
|
||||
|
||||
return norm_depth_blurred, debug_depth_view
|
||||
|
||||
|
||||
def _outpaint_missing_regions(
|
||||
self,
|
||||
img: torch.Tensor,
|
||||
hole_mask: torch.Tensor, # Expects the specific hole_mask
|
||||
outpaint_node: OutpaintAnyProjection,
|
||||
camera_type: str,
|
||||
horizontal_fov: float,
|
||||
width: int,
|
||||
height: int,
|
||||
patch_projection: str,
|
||||
patch_horiz_fov: float,
|
||||
patch_res: int,
|
||||
patch_phi: float,
|
||||
patch_theta: float,
|
||||
prompt: str,
|
||||
num_inference_steps: int,
|
||||
guidance_scale: float,
|
||||
mask_blur: int,
|
||||
) -> torch.Tensor: # Returns only out_img, out_mask is not used later
|
||||
out_img, _ = outpaint_node.outpaint_any( # Assign out_mask to _
|
||||
img,
|
||||
input_projection = camera_type,
|
||||
input_horiz_fov = horizontal_fov,
|
||||
output_projection = camera_type,
|
||||
output_horiz_fov = horizontal_fov,
|
||||
output_width = width,
|
||||
output_height = height,
|
||||
patch_projection = patch_projection,
|
||||
patch_horiz_fov = patch_horiz_fov,
|
||||
patch_res = patch_res,
|
||||
patch_phi = patch_phi,
|
||||
patch_theta = patch_theta,
|
||||
prompt = prompt,
|
||||
num_inference_steps = num_inference_steps,
|
||||
cached = False,
|
||||
guidance_scale = guidance_scale,
|
||||
mask_blur = mask_blur,
|
||||
mask = hole_mask, # Use the passed hole_mask
|
||||
debug = False,
|
||||
)
|
||||
return out_img
|
||||
|
||||
def _prepare_view_data(
|
||||
self,
|
||||
current_enriched_pc: torch.Tensor,
|
||||
M_np: np.ndarray,
|
||||
transform_node: TransformPointCloud,
|
||||
clean_node: PointCloudCleaner,
|
||||
proj_node: ProjectPointCloud,
|
||||
voxel_size: float,
|
||||
min_points_per_voxel: int,
|
||||
camera_type: str,
|
||||
horizontal_fov: float,
|
||||
width: int,
|
||||
height: int,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
# transform and select front points
|
||||
rotated, = transform_node.transform_pointcloud(current_enriched_pc, M_np)
|
||||
pc_front = rotated[rotated[:, 2] > 0]
|
||||
|
||||
# clean front points
|
||||
pc_front, = clean_node.clean_pointcloud(
|
||||
pc_front,
|
||||
voxel_size=voxel_size,
|
||||
min_points_per_voxel=min_points_per_voxel,
|
||||
width=4096, # Consider passing these as params if they vary
|
||||
height=4096, # Consider passing these as params if they vary
|
||||
)
|
||||
|
||||
# project to image + depth
|
||||
img, mask, depth_map = proj_node.project_pointcloud(
|
||||
pc_front,
|
||||
camera_type,
|
||||
horizontal_fov,
|
||||
width,
|
||||
height,
|
||||
point_size=3,
|
||||
return_inverse_depth=False,
|
||||
)
|
||||
return img, mask, depth_map, pc_front
|
||||
pc_world, = transform_node.transform_pointcloud(pc_new, M_inv)
|
||||
# enriched_pc is not rotated
|
||||
enriched_pc = torch.cat([enriched_pc, pc_world.to(device)], dim=0)
|
||||
return enriched_pc, debug_img, norm_depth
|
||||
|
||||
|
||||
NODE_CLASS_MAPPINGS = {"FisheyeDepthEstimator": FisheyeDepthEstimator,
|
||||
"PointcloudTrajectoryEnricher": PointcloudTrajectoryEnricher}
|
||||
|
||||
|
After Width: | Height: | Size: 12 MiB |
|
After Width: | Height: | Size: 224 KiB |
|
After Width: | Height: | Size: 13 MiB |
@@ -0,0 +1,127 @@
|
||||
# LingBot-World 2.0 → 4D video: analysis & integration report
|
||||
|
||||
*Research date: 2026-07-13. LingBot-World 2.0 was released 2026-07-09, four days before this report.*
|
||||
|
||||
## TL;DR
|
||||
|
||||
**LingBot-World 2.0 is not a 3D/4D model — it is a camera-pose- and action-conditioned autoregressive video generator.** It outputs only pixels and maintains no explicit geometry. But it has exactly the property that makes a video-generation model useful for 4D reconstruction: **you command the camera trajectory (poses + intrinsics) of every generated frame**, so every output video is a *posed* video. That turns it into a controllable multi-view video factory whose output can be lifted into 4D Gaussian splats by the existing `video_to_4d_world.json` pipeline in this repo — with the pose-estimation step optionally replaced by the commanded poses.
|
||||
|
||||
Feasibility verdicts:
|
||||
|
||||
| Question | Verdict |
|
||||
| --- | --- |
|
||||
| 4D video from a 3D scene (splat/mesh) | **Yes, indirectly** — render the 3D scene to a seed image, then LingBot animates + explores it. 3D enters only as a rendered start frame; there is no native 3D conditioning. |
|
||||
| 4D Gaussian-splat video from its output | **Feasible and first-party-endorsed** — the LingBot-World paper itself demonstrates reconstructing its generated videos into point clouds with VGGT-class models, the same VGGT this repo already uses. |
|
||||
| Drop-in ComfyUI use today | **Not yet** — 14B Wan2.2-based weights, no quantized release for v2, no wrapper support yet ([kijai/WanVideoWrapper#1920](https://github.com/kijai/ComfyUI-WanVideoWrapper/issues/1920), [Comfy-Org/ComfyUI#12154](https://github.com/Comfy-Org/ComfyUI/issues/12154)); reference inference is 8×GPU `torchrun`. |
|
||||
| Commercial use | **v2: no** (CC BY-NC-SA 4.0). **v1: yes** (Apache 2.0). This alone may decide which version to build on. |
|
||||
|
||||
---
|
||||
|
||||
## 1. What LingBot-World 2.0 actually is
|
||||
|
||||
**Repos & papers**
|
||||
- v2 (current): [Robbyant/lingbot-world-v2](https://github.com/Robbyant/lingbot-world-v2) — "Infinite Worlds with Versatile Interactions", tech report [arXiv:2607.07534](https://arxiv.org/abs/2607.07534), weights [robbyant/lingbot-world-v2-14b-causal-fast](https://huggingface.co/robbyant/lingbot-world-v2-14b-causal-fast). Released 2026-07-09 by Robbyant (embodied-AI subsidiary of Ant Group).
|
||||
- v1 (deprecated but still useful): [Robbyant/lingbot-world](https://github.com/robbyant/lingbot-world) — "Advancing Open-source World Models", [arXiv:2601.20540](https://arxiv.org/abs/2601.20540), weights `robbyant/lingbot-world-base-cam` / `-base-act` / `-fast`. Released 2026-01-29.
|
||||
|
||||
**Architecture (verified against code + paper)**
|
||||
- Built on **Wan2.2 i2v-A14B**: a two-expert MoE video diffusion model, ~28B total parameters with **14B active** per denoising step (high-noise expert for global structure, low-noise for detail). Ships the Wan2.1 VAE and umT5-XXL text encoder.
|
||||
- v2 converts it to **causal, chunk-by-chunk autoregressive generation**: latents are generated `chunk_size` latent frames at a time against a **KV cache** with **sink tokens** and a **local attention window** (`run_fast.sh` uses `--local_attn_size 18 --sink_size 6`). A **MoBA mask** ("Mixture of Bidirectional and Autoregressive Attention Mask") mixes bidirectional attention into teacher forcing to stop the long-horizon quality collapse that plagues autoregressive video. Result: the paper demonstrates an **uninterrupted hour-long session with no perceptible quality decay**.
|
||||
- Two inference modes: `causal_fast` (distilled few-step; drives **720p @ 60 fps** in their real-time deployment) and `causal_pretrain` (40-step CFG; checkpoint still marked TODO). A single-GPU **1.3B variant is described in the paper but not released**.
|
||||
|
||||
**Conditioning inputs — the part that matters for 4D** (from `wan/image2video.py` + `wan/utils/cam_utils.py`)
|
||||
- **Seed image** (`--image`) + **text prompt**: the world is initialized from one image and a background description. This is the *only* way content enters — no 3D input of any kind.
|
||||
- **Camera trajectory**: `poses.npy` `[T,4,4]` **camera-to-world, OpenCV convention** + `intrinsics.npy` `[T,4]` = `[fx,fy,cx,cy]`. Converted to per-pixel **Plücker ray embeddings** (`get_plucker_embeddings`), folded into the latent grid and injected per-chunk into the DiT (AdaLN per the tech report). Relative poses are translation-normalized (`compute_relative_poses`), and `interpolate_camera_poses` (SLERP) is provided.
|
||||
- **Keyboard actions**: `wasd_action.npy` (movement) / `ijkl_action.npy` (view) as multi-hot vectors concatenated onto the Plücker conditioning. v2 adds character actions (attack, archery, spell-cast, shoot, jump, glide) and **chunk-wise text events** (weather, entity spawning, time-of-day), plus a VLM-driven "pilot/director" agentic harness.
|
||||
- v1 README explicitly recommends **[NVIDIA ViPE](https://github.com/nv-tlabs/vipe)** to extract `poses.npy`/`intrinsics.npy` from an *existing real video* — i.e., the official video→control-signal bridge.
|
||||
|
||||
**Inference & hardware**
|
||||
```bash
|
||||
torchrun --nproc_per_node=8 generate.py --task i2v-A14B --size 480*832 \
|
||||
--frame_num 361 --ckpt_dir lingbot-world-v2-14b-causal-fast \
|
||||
--image examples/03/image.jpg --action_path examples/03 \
|
||||
--infer_mode causal_fast --dit_fsdp --t5_fsdp --ulysses_size 8 \
|
||||
--local_attn_size 18 --sink_size 6
|
||||
```
|
||||
- Reference: 8×GPU (FSDP + Ulysses sequence parallel), 480×832, 361 frames (`frame_num` must be 4n+1). Single-GPU runs auto-enable `--offload_model` (T5/DiT swapped to CPU between stages) — expect 80GB-class VRAM for comfortable 14B bf16 inference; there is **no quantized v2 release yet**. v1 has a community **4-bit quant** and `--t5_cpu`, and supports up to 961 frames (~1 min @ 16 fps).
|
||||
- Requirements: `torch >= 2.4.0`, `flash_attn`.
|
||||
|
||||
**License** — v2 code *and* weights are **CC BY-NC-SA 4.0 (non-commercial, share-alike)**; v1 is **Apache 2.0**. Anything commercial built on v2 outputs is off the table; v1 remains the commercially safe option at lower quality/horizon.
|
||||
|
||||
---
|
||||
|
||||
## 2. Can it turn 3D into 4D video?
|
||||
|
||||
**Yes, with the 3D scene entering as a rendered image, not as geometry.** The paper is explicit that the world "is initialized from an initial image and its background description" — there is no splat/mesh/point-cloud conditioning path, and the model "operates without an explicit notion of geometry."
|
||||
|
||||
The working recipe, using nodes already in this repo:
|
||||
|
||||
1. **Render a seed view** of your static 3D asset: `LoadPlySplat` → `RenderSplat` (or a mesh render) at 832×480+, from a pose with good scene coverage.
|
||||
2. **Author the camera trajectory you want** in the splat's own coordinate frame (`CameraInterpolationNode` / `CameraTrajectoryNode`), convert to camera-to-world OpenCV `poses.npy` + `intrinsics.npy`.
|
||||
3. **Feed image + poses + actions/text-events to LingBot-World.** The model animates the scene (wind, characters, weather, spawned entities via text events) while following your camera — i.e., it *invents plausible dynamics* for your static 3D scene. This is "3D → 4D video" in the sense of *generating* the time dimension, not simulating it: physics is learned and imperfect, and the output will drift from your 3D asset's exact geometry the further the camera goes from the seed view.
|
||||
4. **Optionally lift the result back to 4D splats** (section 3) so the animated version of your scene becomes re-renderable from any camera.
|
||||
|
||||
Caveat on fidelity: only the seed frame is constrained by your 3D input. Occluded/unseen regions are hallucinated. For higher fidelity to the source scene you can seed successive generations from renders at multiple poses and stitch — the same strategy `SplatTrajectoryEnricher` already uses with Flux outpainting, but with LingBot providing temporally coherent *video* instead of stills.
|
||||
|
||||
---
|
||||
|
||||
## 3. Feasibility: 4D Gaussian-splat video from LingBot output
|
||||
|
||||
**This is the strongest part of the story.** Three findings, all verified against primary sources:
|
||||
|
||||
1. **Posed video for free.** Because generation is conditioned on `poses.npy`/`intrinsics.npy`, every generated frame comes with a commanded camera. A monocular real video gives you poses only after VGGT/COLMAP estimation; LingBot gives you the trajectory you asked for. (Treat commanded poses as *approximate* — the model follows them but is not geometrically exact; see limitations.)
|
||||
2. **First-party evidence that reconstruction works.** The LingBot-World paper itself demonstrates: *"by leveraging large-scale 3D reconstruction foundation models [lin2025depth, wang2025vggt], we can further convert the generated video sequences into high-quality scene point clouds"*, with point clouds showing *"strong spatial coherence across frames"* (Fig. 16, [arXiv:2601.20540](https://arxiv.org/html/2601.20540v1)). That is literally VGGT — the model behind this repo's `VideoPoseEstimator` — applied to LingBot output by its own authors.
|
||||
3. **Long-horizon consistency is the v2 headline.** Landmarks stay structurally intact after being out of view for up to ~60 s (v1) and v2 extends coherent generation to hour scale with no perceptible decay. Long consistent orbits are exactly what splat optimization needs.
|
||||
|
||||
**How it maps onto known video-to-4D paradigms:**
|
||||
- **CAT4D-style** ([arXiv:2411.18613](https://arxiv.org/abs/2411.18613)): camera/time-disentangled video diffusion → deformable 3DGS optimization. LingBot is not time-disentangled (you cannot freeze time and move the camera — camera and time advance together in one causal stream), so you *cannot* get true simultaneous multi-view of a dynamic instant from a single run.
|
||||
- **Monocular 4D lifting** (this repo's pipeline): works on any single posed video — LingBot output qualifies directly and improves on real footage by letting you *choose* a camera path that orbits/parallaxes around the action, which is the single biggest quality lever for monocular 4D reconstruction.
|
||||
- **Multi-run multi-view**: re-running with the same seed image but different trajectories gives multiple views of the *same static scene* but **different sampled dynamics** (different seeds/action outcomes per run) — usable for static splat fusion, **not** for dynamic 4D supervision. Keep dynamics within one continuous run.
|
||||
|
||||
**Bottom line:** treat LingBot-World as a *trajectory-controllable monocular video source* feeding the existing 4D pipeline; don't expect synchronized multi-view rigs out of it.
|
||||
|
||||
---
|
||||
|
||||
## 4. Concrete pipeline: video → 4D video / 4D splats
|
||||
|
||||
### Path A — real video in, 4D world out, LingBot as the world extender
|
||||
|
||||
Your existing `video_to_4d_world.json` already handles real-video → 4D. LingBot adds value where that pipeline is weakest: viewpoints the source video never saw.
|
||||
|
||||
1. **Base 4D scene from the real video** (existing flow): `VideoPoseEstimator` (VGGT poses/depth) → `ZDepthToRayDepthNode` → `MotionMaskFromDepth` → `VideoToFusedSplats` + `SplatPolish` (static) → `EstimateTracks`/`TracksToTrajectories`/`SplitSplatsByMask`/`BuildSplats4D` (dynamic) → `GSPLAT4D`.
|
||||
2. **Extract control signals from the same video** with ViPE (officially recommended) or reuse the VGGT poses: `VideoPoseEstimator` outputs world-to-camera `[T,4,4]` → `TrajectoryInvert` → camera-to-world OpenCV → export `poses.npy` + `intrinsics.npy` (VGGT's FOV output gives `fx,fy`; `cx,cy` = image center). *(Small new node needed: `TrajectoryToNpyExport` — trivial, ~20 lines.)*
|
||||
3. **Continue the world where the video ends**: last real frame = LingBot seed image; author an exploration trajectory (orbit, dolly, walk) continuing from the last real pose; generate 361+ frames.
|
||||
4. **Lift the generated segment** through the same stage-1 flow and **fuse into the base scene**: `FuseSplats`/`MergeSplats` for statics (scale-anchor with `DepthScaleAnchor` against the base scene's depth), separate `BuildSplats4D` time range for new dynamics. Result: a 4D world larger than the source footage.
|
||||
|
||||
### Path B — single image or 3D scene in, 4D splat video out
|
||||
|
||||
1. **Seed**: any image, or a render of an existing splat (`RenderSplat`) / mesh.
|
||||
2. **Trajectory design**: slow orbit or arc around the subject + gentle forward motion — maximize parallax, avoid pure rotation (no baseline → no geometry). Keep FOV fixed; write `poses.npy`/`intrinsics.npy` (c2w, OpenCV; translations get normalized internally, so keep the trajectory scale moderate and re-anchor metric scale later with `DepthScaleAnchor`).
|
||||
3. **Generate** with `causal_fast`, 480×832, 361 frames; drive dynamics with keyboard/character actions and chunk-wise text events ("a horse gallops through", "rain starts").
|
||||
4. **Reconstruct** — two pose options:
|
||||
- *Trust-but-verify (recommended)*: run `VideoPoseEstimator` on the generated frames anyway; compare with commanded poses (`TrajectoryCompose` of one with `TrajectoryInvert` of the other should be ≈ identity); use VGGT's poses for reconstruction, commanded poses as sanity check. This absorbs the model's camera-following error.
|
||||
- *Fast path*: use commanded poses directly, skip VGGT pose estimation, still run its depth head (or `VideoMetricDepthEstimate`) for the depth maps the lifting nodes need.
|
||||
5. **Lift to 4D**: identical to the existing workflow — motion mask → static fusion (`VideoToFusedSplats` + `SplatPolish`) → tracks (`EstimateTracks` is CoTracker3, works fine on generated footage) → `BuildSplats4D` → `RenderSplats4DVideo` along any novel camera path → `SaveSplats4D`.
|
||||
|
||||
### Integration notes for camera-comfyUI
|
||||
|
||||
- **Coordinate conventions align well**: LingBot uses OpenCV c2w + `[fx,fy,cx,cy]`, this repo's `TRAJECTORY` is 4×4 matrices with `TrajectoryInvert`/`TrajectoryCompose` already available. Needed glue: (a) `TrajectoryToNpyExport` / `NpyToTrajectory` nodes, (b) optionally a `LingBotGenerate` node wrapping `generate.py` via subprocess for remote/8-GPU boxes — running 14B in-process inside ComfyUI is not realistic today.
|
||||
- **ComfyUI-native inference isn't there yet**: WanVideoWrapper/ComfyUI support for LingBot checkpoints is an open request blocked on VRAM/quantization ([#1920](https://github.com/kijai/ComfyUI-WanVideoWrapper/issues/1920), [#12154](https://github.com/Comfy-Org/ComfyUI/issues/12154)). Because it's Wan2.2-architecture, wrapper support and GGUF/FP8 quants are likely to appear quickly; the causal KV-cache/sink/MoBA inference loop is custom, so a naive Wan2.2 loader won't reproduce long-horizon behavior.
|
||||
- **Pragmatic hardware ladder**: (1) today, single-image experiments on v1 `base-cam` 4-bit quant (Apache 2.0, 480p, camera-pose conditioned — same poses.npy interface) on a 24 GB GPU; (2) v2 14B on a rented 8×A100/H100 node or single 80 GB GPU with offload; (3) wait for the announced 1.3B v2 release for true single-GPU local use.
|
||||
|
||||
### Known limitations
|
||||
|
||||
- **No geometry inside the model** — all 3D/4D structure comes from post-hoc reconstruction; physics is "imperfect" by the authors' own admission.
|
||||
- **Camera-following error**: commanded poses ≠ achieved poses exactly (Plücker conditioning is a soft constraint; translations are normalized, so absolute scale is undefined) — always re-anchor scale and consider re-estimating poses.
|
||||
- **Dynamics are not repeatable across runs** — multi-view supervision of a dynamic instant is impossible; design single continuous runs whose camera moves *around* the action.
|
||||
- **480×832 native offline resolution** (720p is the real-time streaming mode) — plan on splat-space upscaling or `SplatPolish` against upscaled frames.
|
||||
- **Generated-content artifacts** (texture shimmer, occasional object morphing) become floaters/ghosts in splat space — the existing `MotionMaskFromDepth` + `DepthEdgeFilter` + `PointCloudCleaner` stack mitigates this, and track-validity filtering in `TracksToTrajectories` matters more than with real footage.
|
||||
- **License**: v2 is CC BY-NC-SA 4.0 — non-commercial only, share-alike. Use v1 (Apache 2.0) for anything with commercial intent.
|
||||
|
||||
---
|
||||
|
||||
## Sources
|
||||
|
||||
Primary: [lingbot-world-v2 repo](https://github.com/Robbyant/lingbot-world-v2) · [v2 tech report arXiv:2607.07534](https://arxiv.org/abs/2607.07534) · [v2 weights (HF)](https://huggingface.co/robbyant/lingbot-world-v2-14b-causal-fast) · [lingbot-world v1 repo](https://github.com/robbyant/lingbot-world) · [v1 paper arXiv:2601.20540](https://arxiv.org/abs/2601.20540) · [v1 cam weights (HF)](https://huggingface.co/robbyant/lingbot-world-base-cam) · code files `generate.py`, `wan/image2video.py`, `wan/utils/cam_utils.py`, `run_fast.sh` (read directly).
|
||||
Secondary: [Robbyant press release (2026-07-09)](https://www.businesswire.com/news/home/20260708757367/en/Robbyant-Unveils-LingBot-World-2.0-Pioneering-Hour-Long-Real-Time-Generation-in-World-Models) · [v1 release (2026-01-28)](https://www.businesswire.com/news/home/20260128459962/en/Robbyant-Open-Sources-LingBot-World-a-World-Model-for-Millisecond-Level-Real-Time-Interaction) · [CAT4D arXiv:2411.18613](https://arxiv.org/abs/2411.18613) · [ViPE](https://github.com/nv-tlabs/vipe) · ComfyUI support threads [WanVideoWrapper#1920](https://github.com/kijai/ComfyUI-WanVideoWrapper/issues/1920), [ComfyUI#12154](https://github.com/Comfy-Org/ComfyUI/issues/12154).
|
||||
|
||||
*Method note: claims were gathered by a fan-out research pass (18 sources, 90 raw claims, 25 adversarially verified: 14 confirmed 3-0, 3 refuted, 8 verification-errored) plus direct reading of both repos' inference code and both arXiv papers. The two load-bearing claims whose automated verification errored (v1's video→point-cloud demonstration; the unreleased 1.3B variant) were re-verified manually against the arXiv HTML.*
|
||||
@@ -0,0 +1,96 @@
|
||||
# Workflow review — July 2026
|
||||
|
||||
Scope: all 17 `workflows/*.json`. Tooling used (kept in `notebooks/`):
|
||||
|
||||
* `validate_workflows.py` — checks every stored `widgets_values` array against
|
||||
the *current* `INPUT_TYPES` of the node it targets (count + combo values).
|
||||
Run it whenever a node's inputs change.
|
||||
* `rework_workflows_2026_07.py` — the one-shot migration that produced the
|
||||
current state of the files (documented below, idempotent-ish).
|
||||
|
||||
## What was wrong (now fixed)
|
||||
|
||||
ComfyUI applies `widgets_values` **positionally**. When a node gains/loses/
|
||||
reorders widgets, old workflows load with silently shifted values — no error,
|
||||
just wrong settings. The validator found 28 such cases across 9 files:
|
||||
|
||||
| Node | Old → new widgets | Files affected | Migration applied |
|
||||
| --- | --- | --- | --- |
|
||||
| `DepthEstimatorNode` | 2 → 3 (`median_blur_kernel` added) | 5 files, 12 nodes | appended default `1` |
|
||||
| `FisheyeDepthEstimator` | 6 → 8 (`mode`, `median_blur_kernel` added) | 2 files | inserted `SOFTMERGE` (radius was already configured), appended `1` |
|
||||
| `PointCloudCleaner` | 2 → 4 (screen-space `width`/`height` added; units changed) | `PointCloud.json` | reset to defaults `[1024, 1024, 1.0, 3]` — old world-unit values were meaningless in the new schema (**retune on GPU box**) |
|
||||
| `CameraMotionNode` | 6 → 9 (`widen_mask`, `invert_mask`, `points_to_mask`) | 3 files | appended defaults `[0, false, false]` |
|
||||
| `CameraInterpolationNode` | 0 → 1 (`num_steps`) | 3 files | set `2` (keyframes only — `CameraMotionNode`/`VideoCameraMotionSequence` interpolate frames themselves) |
|
||||
| `PointcloudTrajectoryEnricher` | 20 → 16 (render/reproject-back options internalized) | `PC_enricher.json` | first 13 kept 1:1, new tail set to defaults |
|
||||
|
||||
Beyond widget drift:
|
||||
|
||||
* **`outpainting_fisheye.json` was structurally broken**: its seven
|
||||
`ReprojectImage` nodes stored patch rotations (±42°) in a pre-2025 widget
|
||||
slot that now lands on the `inverse` boolean (42 → truthy). Repaired by
|
||||
adding two `TransformToMatrix` nodes (±42°) wired to `transform_matrix`
|
||||
inputs and setting proper `inverse` flags — mirroring the structure of
|
||||
`outpainting_fisheye_flux.json`.
|
||||
* **`outpainting_fisheye_flux.json`** stored `45` in one `inverse` slot
|
||||
(truthy-by-accident); normalized all seven to real booleans.
|
||||
* **`wan_vace_ref_to_video.json`** pointed the UNETLoader at
|
||||
`wan2,1_vace14B_fp16.safetensors` (comma typo + missing underscore) — no such
|
||||
file can exist; fixed to `wan2.1_vace_14B_fp16.safetensors` (matches
|
||||
`install.sh`'s download name).
|
||||
* `Pointcloud walker.json` / `Test pointcloud_loading.json` renamed to
|
||||
`pointcloud_walker.json` / `test_pointcloud_loading.json` (spaces break
|
||||
shell ergonomics and URL linking).
|
||||
* README workflow table was stale (claimed `pointcloud_walker` was an "Open3D
|
||||
GUI", missed 5 workflows); rewritten with per-workflow extras columns.
|
||||
|
||||
Every workflow now carries an embedded **“About this workflow”** MarkdownNote
|
||||
(purpose, stages, what to set, required packs/models), meaningful group boxes,
|
||||
and titles on the nodes users are expected to edit.
|
||||
|
||||
## Improvements — applied 2026-07-16
|
||||
|
||||
Implemented by `notebooks/apply_improvements_2026_07.py` plus repo changes:
|
||||
|
||||
1. **Runnable defaults** ✅ — `example_inputs/` ships
|
||||
`camera_example_pinhole.jpg`, `camera_example_fisheye.jpg` and
|
||||
`ComfyUITrajectory_00001.npy`; `install.py` copies them into ComfyUI's
|
||||
`input/` dir (no overwrite), and every LoadImage/LoadTrajectory default
|
||||
points at them. Exception: `video_camera.json` still needs a user video
|
||||
(none bundled — a clip would bloat the archive).
|
||||
2. **Template browser** ✅ — `workflows/` is already an accepted template
|
||||
directory name (per docs.comfy.org, alongside `example_workflows`), so no
|
||||
rename was needed; added the missing same-name `.jpg` thumbnails (10
|
||||
workflows, generated 512 px from `demo_images/`) and removed the stray
|
||||
`workflows/__init__.py`. Legacy graphs moved to `workflows/legacy/`, which
|
||||
keeps them out of the browser.
|
||||
3. **`video_camera.json` dep trim** ✅ — removed the dead-end `BlurMaskFast`
|
||||
(blur radius was 0/0 — a no-op even if wired), the `easy mathInt` pad
|
||||
computation and the then-dangling `VHS_VideoInfo`; pad top/bottom now use
|
||||
the node's stored values (set to `(W−H)/2`, note explains); swapped
|
||||
KJNodes' `GetImageRangeFromBatch` for the built-in `ImageFromBatch`.
|
||||
Remaining pack deps: VideoHelperSuite + Florence2 (was five packs).
|
||||
4. **Fisheye-outpaint consolidation** ✅ — SD-checkpoint variant moved to
|
||||
`workflows/legacy/outpainting_fisheye.json`.
|
||||
5. **Depth consolidation** ✅ — `workflows/legacy/Fisheye_depth_workflow.json`.
|
||||
6. **Trajectory recorder** ✅ — new `workflows/record_trajectory.json`
|
||||
(TransformToMatrix ×2 → CameraInterpolationNode → SaveTrajectory), and the
|
||||
bundled example trajectory covers the zero-setup path.
|
||||
7. **CI guard** ✅ — `.github/workflows/validate.yml` runs the workflow
|
||||
validator, installer-logic tests and the 4D smoke suite on CPU torch for
|
||||
every PR and push to main.
|
||||
8. **Schema versioning** ✅ — every workflow now carries
|
||||
`extra.camera_comfyui_rev = 1`; bump on the next migration.
|
||||
9. Bonus: a bidirectional link-integrity sweep found and pruned 8 stale link
|
||||
references (pre-existing) plus one dangling `mask` input in
|
||||
`outpainting_fisheye_flux.json`.
|
||||
|
||||
## Still open
|
||||
|
||||
* **Retune migrated defaults on the GPU box.** `PointCloudCleaner` (in
|
||||
`PointCloud.json`) and the voxel-merge tail of `PointcloudTrajectoryEnricher`
|
||||
(in `PC_enricher.json`) were reset to schema defaults; verify visual quality
|
||||
and bake in good values.
|
||||
* **Load-and-queue pass in real ComfyUI.** All checks here are static; open
|
||||
each template once on the GPU box to confirm layout and execution.
|
||||
* **Bundle a small example video** for `video_camera.json` if archive size
|
||||
allows (or document a public sample clip URL in its note).
|
||||
|
After Width: | Height: | Size: 1.3 MiB |
|
After Width: | Height: | Size: 196 KiB |
@@ -4,16 +4,52 @@ import numpy as np
|
||||
# reprojection helpers
|
||||
from .reprojection_nodes import Projection, ReprojectImage, TransformToMatrix
|
||||
|
||||
# Try importing FluxInpainting and capture any ImportError
|
||||
import importlib
|
||||
import sys, os, logging
|
||||
here = os.path.dirname(os.path.dirname(os.path.dirname(__file__)))
|
||||
if here not in sys.path:
|
||||
sys.path.append(here)
|
||||
|
||||
# Folder names the ComfyUI-Flux-Inpainting pack may live under in custom_nodes/
|
||||
# (install.py clones it as "inpainting_flux"; keep both lists in sync).
|
||||
_FLUX_PACK_CANDIDATES = (
|
||||
"inpainting_flux",
|
||||
"ComfyUI-Flux-Inpainting",
|
||||
"ComfyUI-Flux-Inpainting-main",
|
||||
"comfyui-flux-inpainting",
|
||||
)
|
||||
|
||||
|
||||
def _import_flux_inpainting():
|
||||
"""Import FluxNF4Inpainting from the flux inpainting pack regardless of the
|
||||
folder name it was installed under."""
|
||||
custom_nodes_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
comfy_root = os.path.dirname(custom_nodes_dir)
|
||||
if comfy_root not in sys.path:
|
||||
sys.path.append(comfy_root)
|
||||
|
||||
last_error = None
|
||||
for name in _FLUX_PACK_CANDIDATES:
|
||||
if not os.path.isfile(os.path.join(custom_nodes_dir, name, "nodes.py")):
|
||||
continue
|
||||
try:
|
||||
module = importlib.import_module(f"custom_nodes.{name}.nodes")
|
||||
return module.FluxNF4Inpainting
|
||||
except Exception as exc:
|
||||
last_error = exc
|
||||
if last_error is None:
|
||||
last_error = ModuleNotFoundError(
|
||||
"No flux inpainting pack found in custom_nodes (looked for "
|
||||
f"{', '.join(_FLUX_PACK_CANDIDATES)}). It is installed automatically "
|
||||
"by this pack's install.py (run by ComfyUI-Manager); to set it up "
|
||||
f"manually, clone {'https://github.com/rubi-du/ComfyUI-Flux-Inpainting'} "
|
||||
"into custom_nodes/inpainting_flux."
|
||||
)
|
||||
raise last_error
|
||||
|
||||
|
||||
# Try importing FluxInpainting and capture any error
|
||||
try:
|
||||
from custom_nodes.inpainting_flux.nodes import FluxNF4Inpainting as FluxInpainting
|
||||
FluxInpainting = _import_flux_inpainting()
|
||||
_flux_import_error = None
|
||||
except ImportError as e:
|
||||
except Exception as e:
|
||||
FluxInpainting = None
|
||||
_flux_import_error = e
|
||||
logging.error(f"[OutpaintAnyProjection] could not import FluxNF4Inpainting: {e}")
|
||||
@@ -71,7 +107,8 @@ class OutpaintAnyProjection:
|
||||
if _flux_import_error is not None:
|
||||
raise RuntimeError(
|
||||
f"FluxNF4Inpainting is not available: {_flux_import_error}\n"
|
||||
"Please install or fix your inpainting_flux package."
|
||||
"Run this pack's install.py (ComfyUI-Manager does this automatically) "
|
||||
"or install/fix custom_nodes/inpainting_flux."
|
||||
)
|
||||
|
||||
def normalize_mask(m: torch.Tensor):
|
||||
@@ -129,7 +166,7 @@ class OutpaintAnyProjection:
|
||||
patch_mask = torch.ones_like(patch_mask) * 1 if debug else patch_mask
|
||||
# 5) Reproject inpainted patch back
|
||||
|
||||
back_img, inpainted_patch_reproj_mask_raw = reproj.reproject_image( # Store raw mask output
|
||||
back_img, back_mask = reproj.reproject_image(
|
||||
inpainted_patch,
|
||||
patch_horiz_fov, output_horiz_fov,
|
||||
patch_projection, output_projection,
|
||||
@@ -138,37 +175,19 @@ class OutpaintAnyProjection:
|
||||
transform_matrix=rot_m,
|
||||
feathering=0,
|
||||
)
|
||||
# inpainted_patch_coverage_mask: 1.0 where the reprojected inpainted patch has content, 0.0 otherwise.
|
||||
inpainted_patch_coverage_mask = normalize_mask(back_mask) # Ensure it's float [0,1]
|
||||
back_mask = normalize_mask(back_mask).bool() # True where patch contributes
|
||||
|
||||
# --- Define Masks based on Conventions ---
|
||||
# base_mask: 1.0 where original reprojected image has content, 0.0 for holes.
|
||||
# initial_hole_mask: 1.0 where original reprojected image has holes (inverse of base_mask).
|
||||
initial_hole_mask = 1.0 - base_mask
|
||||
# inpainted_patch_coverage_mask: 1.0 where reprojected inpainted patch has content.
|
||||
# original coverage: True = had data, False = hole
|
||||
orig_covered = ~base_mask.bool()
|
||||
base_img=base_img * orig_covered.unsqueeze(-1)
|
||||
# fill only holes where back_mask is False
|
||||
filled = back_img * (~back_mask.unsqueeze(-1))*base_mask.unsqueeze(-1)
|
||||
final_img = base_img+filled
|
||||
|
||||
# --- Compositing Logic ---
|
||||
# Goal: Inpainted patch takes precedence in overlapping areas. Original content is used elsewhere.
|
||||
# anything that’s still a hole after back‐projection needs inpaint
|
||||
needs_inpaint = (~((orig_covered) | (~back_mask))).to(torch.float32)
|
||||
|
||||
# Contribution from the original image:
|
||||
# Valid original pixels, excluding areas covered by the inpainted patch.
|
||||
original_content_contribution = base_img * base_mask.unsqueeze(-1) * \
|
||||
(1.0 - inpainted_patch_coverage_mask.unsqueeze(-1))
|
||||
|
||||
# Contribution from the inpainted patch (reprojected as back_img):
|
||||
# Valid inpainted pixels, where the patch provides coverage.
|
||||
inpainted_patch_contribution = back_img * inpainted_patch_coverage_mask.unsqueeze(-1)
|
||||
|
||||
# Combine:
|
||||
final_img = original_content_contribution + inpainted_patch_contribution
|
||||
|
||||
# --- needs_inpaint_mask Derivation ---
|
||||
# Identifies areas that were initially holes AND remain un-filled by the reprojected inpainted patch.
|
||||
# These are areas that still require inpainting if a further pass was to be made.
|
||||
not_covered_by_inpainted_patch = 1.0 - inpainted_patch_coverage_mask
|
||||
needs_inpaint_mask = initial_hole_mask * not_covered_by_inpainted_patch # Element-wise multiplication (AND logic)
|
||||
|
||||
return final_img, needs_inpaint_mask
|
||||
return final_img, needs_inpaint
|
||||
|
||||
# register
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
|
||||
@@ -0,0 +1,215 @@
|
||||
"""Post-install setup for camera-comfyUI.
|
||||
|
||||
ComfyUI-Manager runs this script automatically after installing the node pack
|
||||
(both git and registry installs), right after `pip install -r requirements.txt`.
|
||||
Manual users can run it themselves: `python install.py` from this directory.
|
||||
|
||||
It makes the optional heavy dependencies work out of the box:
|
||||
* vggt — pip-installed from GitHub (not on PyPI); needed by VideoPoseEstimator.
|
||||
* SHARP — the submodules/ml-sharpt checkout; needed by ImageToSplat & co.
|
||||
* gsplat — SHARP's rasterizer backend (JIT-compiles CUDA kernels on first use).
|
||||
* inpainting_flux — the ComfyUI-Flux-Inpainting node pack, cloned as a sibling
|
||||
custom node; needed by OutpaintAnyProjection / SplatTrajectoryEnricher.
|
||||
|
||||
Every step is idempotent and non-fatal: the node pack degrades gracefully at
|
||||
runtime (see __init__.py), so a failed optional step only disables the nodes
|
||||
that need it. The script therefore always exits 0 and reports what it skipped.
|
||||
"""
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
NODE_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
LOG_PREFIX = "[camera-comfyUI install]"
|
||||
|
||||
VGGT_GIT_URL = "https://github.com/facebookresearch/vggt.git"
|
||||
SHARP_GIT_URL = "https://github.com/apple/ml-sharp"
|
||||
FLUX_PACK_GIT_URL = "https://github.com/rubi-du/ComfyUI-Flux-Inpainting.git"
|
||||
# Folder names under custom_nodes/ that count as "the flux inpainting pack is
|
||||
# already installed" (must stay in sync with flux_fisheye_filling_nodes.py).
|
||||
FLUX_PACK_CANDIDATES = (
|
||||
"inpainting_flux",
|
||||
"ComfyUI-Flux-Inpainting",
|
||||
"ComfyUI-Flux-Inpainting-main",
|
||||
"comfyui-flux-inpainting",
|
||||
)
|
||||
|
||||
|
||||
def _log(message: str) -> None:
|
||||
print(f"{LOG_PREFIX} {message}", flush=True)
|
||||
|
||||
|
||||
def _run(cmd, cwd=None) -> None:
|
||||
_log("$ " + " ".join(cmd))
|
||||
subprocess.check_call(cmd, cwd=cwd)
|
||||
|
||||
|
||||
def _pip_install(*args: str) -> None:
|
||||
_run([sys.executable, "-m", "pip", "install", *args])
|
||||
|
||||
|
||||
def _git() -> str:
|
||||
git = shutil.which("git")
|
||||
if git is None:
|
||||
raise RuntimeError(
|
||||
"git executable not found on PATH; cannot fetch GitHub dependencies"
|
||||
)
|
||||
return git
|
||||
|
||||
|
||||
def _importable(module_name: str) -> bool:
|
||||
import importlib.util
|
||||
|
||||
try:
|
||||
return importlib.util.find_spec(module_name) is not None
|
||||
except (ImportError, ValueError):
|
||||
return False
|
||||
|
||||
|
||||
def ensure_base_requirements() -> None:
|
||||
"""Install requirements.txt if it clearly has not been installed yet.
|
||||
|
||||
ComfyUI-Manager installs it before running this script, so this only fires
|
||||
for manual `python install.py` users.
|
||||
"""
|
||||
if _importable("diffusers") and _importable("transformers"):
|
||||
_log("base requirements already satisfied")
|
||||
return
|
||||
_pip_install("-r", os.path.join(NODE_DIR, "requirements.txt"))
|
||||
|
||||
|
||||
def ensure_vggt() -> None:
|
||||
"""VGGT (VideoPoseEstimator). Not on PyPI — install from GitHub over https."""
|
||||
if _importable("vggt"):
|
||||
_log("vggt already installed")
|
||||
return
|
||||
_pip_install(f"vggt @ git+{VGGT_GIT_URL}")
|
||||
_log("vggt installed from GitHub")
|
||||
|
||||
|
||||
def ensure_sharp_checkout() -> None:
|
||||
"""Materialize the SHARP submodule (ImageToSplat / VideoToFusedSplats).
|
||||
|
||||
Registry archives already bundle it; git installs need `submodule update`
|
||||
(ComfyUI-Manager does not clone recursively). If this directory is not a
|
||||
git checkout at all, fall back to a direct clone.
|
||||
"""
|
||||
sharp_dir = os.path.join(NODE_DIR, "submodules", "ml-sharpt")
|
||||
sharp_src = os.path.join(sharp_dir, "src", "sharp")
|
||||
if os.path.isdir(sharp_src):
|
||||
_log("SHARP checkout already present")
|
||||
return
|
||||
|
||||
git = _git()
|
||||
if os.path.exists(os.path.join(NODE_DIR, ".git")):
|
||||
_run([git, "submodule", "update", "--init", "--recursive"], cwd=NODE_DIR)
|
||||
if os.path.isdir(sharp_src):
|
||||
_log("SHARP submodule initialized")
|
||||
return
|
||||
|
||||
if os.path.isdir(sharp_dir) and os.listdir(sharp_dir):
|
||||
raise RuntimeError(
|
||||
f"{sharp_dir} exists but does not contain src/sharp; "
|
||||
"remove it and re-run install.py"
|
||||
)
|
||||
_run([git, "clone", "--depth", "1", SHARP_GIT_URL, sharp_dir])
|
||||
_log("SHARP cloned from GitHub")
|
||||
|
||||
|
||||
def ensure_gsplat() -> None:
|
||||
"""gsplat backs SHARP's import chain and the fast splat render path.
|
||||
|
||||
pip install is lightweight (CUDA kernels JIT-compile on first use), but it
|
||||
can still fail on exotic setups — that only disables the SHARP/gsplat nodes.
|
||||
"""
|
||||
if _importable("gsplat"):
|
||||
_log("gsplat already installed")
|
||||
return
|
||||
_pip_install("gsplat")
|
||||
_log("gsplat installed")
|
||||
|
||||
|
||||
def ensure_flux_inpainting_pack() -> None:
|
||||
"""Clone ComfyUI-Flux-Inpainting next to this pack if no copy exists yet."""
|
||||
custom_nodes_dir = os.path.dirname(NODE_DIR)
|
||||
if os.path.basename(custom_nodes_dir).lower() != "custom_nodes":
|
||||
_log(
|
||||
"not installed under a ComfyUI custom_nodes directory; "
|
||||
"skipping ComfyUI-Flux-Inpainting setup"
|
||||
)
|
||||
return
|
||||
|
||||
for name in FLUX_PACK_CANDIDATES:
|
||||
if os.path.isdir(os.path.join(custom_nodes_dir, name)):
|
||||
_log(f"flux inpainting pack already present ({name})")
|
||||
return
|
||||
|
||||
git = _git()
|
||||
target = os.path.join(custom_nodes_dir, "inpainting_flux")
|
||||
_run([git, "clone", "--depth", "1", FLUX_PACK_GIT_URL, target])
|
||||
pack_requirements = os.path.join(target, "requirements.txt")
|
||||
if os.path.isfile(pack_requirements):
|
||||
_pip_install("-r", pack_requirements)
|
||||
_log("ComfyUI-Flux-Inpainting installed as custom_nodes/inpainting_flux")
|
||||
|
||||
|
||||
def ensure_example_inputs() -> None:
|
||||
"""Copy bundled example inputs into ComfyUI's input dir (no overwrite).
|
||||
|
||||
The shipped example workflows reference these files, so a fresh install
|
||||
can queue them immediately.
|
||||
"""
|
||||
src_dir = os.path.join(NODE_DIR, "example_inputs")
|
||||
if not os.path.isdir(src_dir):
|
||||
_log("no example_inputs directory; skipping")
|
||||
return
|
||||
custom_nodes_dir = os.path.dirname(NODE_DIR)
|
||||
if os.path.basename(custom_nodes_dir).lower() != "custom_nodes":
|
||||
_log("not installed under a ComfyUI custom_nodes directory; "
|
||||
"skipping example input setup")
|
||||
return
|
||||
input_dir = os.path.join(os.path.dirname(custom_nodes_dir), "input")
|
||||
os.makedirs(input_dir, exist_ok=True)
|
||||
copied = 0
|
||||
for name in os.listdir(src_dir):
|
||||
target = os.path.join(input_dir, name)
|
||||
if not os.path.exists(target):
|
||||
shutil.copy2(os.path.join(src_dir, name), target)
|
||||
copied += 1
|
||||
_log(f"example inputs ready ({copied} file(s) copied to {input_dir})")
|
||||
|
||||
|
||||
STEPS = (
|
||||
("base requirements", ensure_base_requirements),
|
||||
("vggt (VideoPoseEstimator)", ensure_vggt),
|
||||
("SHARP checkout (ImageToSplat)", ensure_sharp_checkout),
|
||||
("gsplat (splat rasterizer)", ensure_gsplat),
|
||||
("ComfyUI-Flux-Inpainting (OutpaintAnyProjection)", ensure_flux_inpainting_pack),
|
||||
("example workflow inputs", ensure_example_inputs),
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
failures = []
|
||||
for title, step in STEPS:
|
||||
_log(f"--- {title} ---")
|
||||
try:
|
||||
step()
|
||||
except Exception as exc: # keep going: each dependency is optional
|
||||
failures.append((title, exc))
|
||||
_log(f"WARNING: {title} failed: {exc}")
|
||||
|
||||
if failures:
|
||||
_log("finished with warnings - the affected optional nodes stay disabled:")
|
||||
for title, exc in failures:
|
||||
_log(f" * {title}: {exc}")
|
||||
_log("re-run `python install.py` after fixing the issue (network/git/pip)")
|
||||
else:
|
||||
_log("all optional dependencies are ready")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,163 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
# ----------------------------------------
|
||||
# Functions
|
||||
# ----------------------------------------
|
||||
install_pytorch() {
|
||||
echo "==> Installing PyTorch, TorchVision, TorchAudio, bitsandbytes, accelerate…"
|
||||
pip3 install -U torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu128
|
||||
pip3 install -U bitsandbytes accelerate
|
||||
}
|
||||
|
||||
install_system_deps() {
|
||||
echo "==> Updating apt and installing system packages…"
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y build-essential ffmpeg libsm6 libxext6 python3.10-dev
|
||||
}
|
||||
|
||||
clone_and_install_comfyui() {
|
||||
echo "==> Cloning ComfyUI…"
|
||||
git clone https://github.com/comfyanonymous/ComfyUI.git
|
||||
echo "==> Installing ComfyUI Python requirements…"
|
||||
pip3 install -r ComfyUI/requirements.txt
|
||||
}
|
||||
|
||||
install_camera_node() {
|
||||
echo "==> Installing camera‑ComfyUI node…"
|
||||
mkdir -p ComfyUI/custom_nodes
|
||||
git clone https://github.com/Alexankharin/camera-comfyUI.git \
|
||||
ComfyUI/custom_nodes/camera-comfyUI
|
||||
pip3 install -r ComfyUI/custom_nodes/camera-comfyUI/requirements.txt
|
||||
# install.py sets up vggt, the SHARP submodule, gsplat and the
|
||||
# inpainting_flux sibling pack (same script ComfyUI-Manager runs).
|
||||
( cd ComfyUI/custom_nodes/camera-comfyUI && python3 install.py )
|
||||
}
|
||||
|
||||
install_image_filters() {
|
||||
echo "==> Installing Image‑Filters node…"
|
||||
mkdir -p ComfyUI/custom_nodes
|
||||
git clone https://github.com/spacepxl/ComfyUI-Image-Filters.git \
|
||||
ComfyUI/custom_nodes/ComfyUI-Image-Filters
|
||||
pip3 install -r ComfyUI/custom_nodes/ComfyUI-Image-Filters/requirements.txt
|
||||
}
|
||||
|
||||
clone_flux_inpainting() {
|
||||
echo "==> Installing Flux‑Inpainting node…"
|
||||
mkdir -p ComfyUI/custom_nodes
|
||||
git clone https://github.com/rubi-du/ComfyUI-Flux-Inpainting.git \
|
||||
ComfyUI/custom_nodes/inpainting_flux
|
||||
}
|
||||
|
||||
install_metric_video_depth_anything() {
|
||||
echo "==> Installing Metric Video Depth Anything…"
|
||||
# Clone into ComfyUI root, not custom_nodes
|
||||
git clone https://github.com/DepthAnything/Video-Depth-Anything.git \
|
||||
ComfyUI/Video-Depth-Anything
|
||||
|
||||
echo " • Installing easydict…"
|
||||
pip3 install -U easydict
|
||||
|
||||
echo " • Copying util.py…"
|
||||
mkdir -p ComfyUI/utils
|
||||
cp ComfyUI/Video-Depth-Anything/metric_depth/utils/util.py \
|
||||
ComfyUI/utils/util.py
|
||||
|
||||
echo " • Downloading Metric Video Depth checkpoint…"
|
||||
mkdir -p ComfyUI/models/checkpoints
|
||||
wget -q -O ComfyUI/models/checkpoints/metric_video_depth_anything_vitl.pth \
|
||||
"https://huggingface.co/depth-anything/Metric-Video-Depth-Anything-Large/resolve/main/metric_video_depth_anything_vitl.pth"
|
||||
}
|
||||
|
||||
|
||||
install_comfyui_manager() {
|
||||
echo "==> Installing ComfyUI-Manager extension…"
|
||||
mkdir -p ComfyUI/custom_nodes
|
||||
git clone https://github.com/Comfy-Org/ComfyUI-Manager.git \
|
||||
ComfyUI/custom_nodes/ComfyUI-Manager
|
||||
pip3 install -r ComfyUI/custom_nodes/ComfyUI-Manager/requirements.txt
|
||||
}
|
||||
|
||||
install_hf_hub() {
|
||||
echo "==> Installing huggingface_hub…"
|
||||
pip3 install -U huggingface_hub
|
||||
}
|
||||
|
||||
download_vae_models() {
|
||||
echo "==> Downloading WAN‑VACE models…"
|
||||
mkdir -p ComfyUI/models/vae
|
||||
wget -q -O ComfyUI/models/vae/wan_2.1_vae.safetensors \
|
||||
"https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/vae/wan_2.1_vae.safetensors?download=true"
|
||||
|
||||
mkdir -p ComfyUI/models/text_encoders
|
||||
wget -q -O ComfyUI/models/text_encoders/umt5_xxl_fp16.safetensors \
|
||||
"https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/text_encoders/umt5_xxl_fp16.safetensors?download=true"
|
||||
|
||||
mkdir -p ComfyUI/models/diffusion_models
|
||||
wget -q -O ComfyUI/models/diffusion_models/wan2.1_vace_14B_fp16.safetensors \
|
||||
"https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/diffusion_models/wan2.1_vace_14B_fp16.safetensors"
|
||||
}
|
||||
|
||||
login_hf_hub() {
|
||||
echo "==> Hugging Face login…"
|
||||
huggingface-cli login
|
||||
}
|
||||
|
||||
# ----------------------------------------
|
||||
# Main
|
||||
# ----------------------------------------
|
||||
MODE="${1:-install}"
|
||||
|
||||
case "$MODE" in
|
||||
modules)
|
||||
install_pytorch
|
||||
install_system_deps
|
||||
clone_and_install_comfyui
|
||||
install_camera_node
|
||||
install_image_filters
|
||||
install_comfyui_manager
|
||||
install_hf_hub
|
||||
;;
|
||||
|
||||
flux)
|
||||
clone_flux_inpainting
|
||||
;;
|
||||
|
||||
vae)
|
||||
download_vae_models
|
||||
;;
|
||||
|
||||
depth)
|
||||
install_metric_video_depth_anything
|
||||
;;
|
||||
|
||||
install)
|
||||
install_pytorch
|
||||
install_system_deps
|
||||
clone_and_install_comfyui
|
||||
install_camera_node
|
||||
install_image_filters
|
||||
install_comfyui_manager
|
||||
install_hf_hub
|
||||
install_metric_video_depth_anything
|
||||
;;
|
||||
|
||||
all)
|
||||
install_pytorch
|
||||
install_system_deps
|
||||
clone_and_install_comfyui
|
||||
install_camera_node
|
||||
install_image_filters
|
||||
install_comfyui_manager
|
||||
install_hf_hub
|
||||
download_vae_models
|
||||
install_metric_video_depth_anything
|
||||
login_hf_hub
|
||||
echo "✅ All done!"
|
||||
;;
|
||||
|
||||
*)
|
||||
echo "Usage: $0 {install|modules|flux|vae|depth|all}"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
@@ -41,7 +41,7 @@ class DepthEstimatorNode:
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("depth tensor",)
|
||||
FUNCTION = "estimate_depth"
|
||||
CATEGORY = "Camera/depth"
|
||||
CATEGORY = "Camera/Depth"
|
||||
|
||||
def estimate_depth(
|
||||
self,
|
||||
@@ -110,7 +110,7 @@ class DepthToImageNode:
|
||||
RETURN_TYPES = ("IMAGE",)
|
||||
RETURN_NAMES = ("depth image",)
|
||||
FUNCTION = "depth_to_image"
|
||||
CATEGORY = "Camera/depth"
|
||||
CATEGORY = "Camera/Depth"
|
||||
|
||||
def depth_to_image(
|
||||
self,
|
||||
@@ -158,7 +158,7 @@ class ZDepthToRayDepthNode:
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("ray depth",)
|
||||
FUNCTION = "depth_to_ray_depth"
|
||||
CATEGORY = "Camera/depth"
|
||||
CATEGORY = "Camera/Depth"
|
||||
|
||||
def depth_to_ray_depth(
|
||||
self,
|
||||
@@ -229,7 +229,7 @@ class CombineDepthsNode:
|
||||
RETURN_TYPES = ("TENSOR","MASK")
|
||||
RETURN_NAMES = ("combined_depth","combined_mask")
|
||||
FUNCTION = "combine_depths"
|
||||
CATEGORY = "Camera/depth"
|
||||
CATEGORY = "Camera/Depth"
|
||||
|
||||
def combine_depths(
|
||||
self,
|
||||
@@ -358,7 +358,7 @@ class DepthRenormalizer:
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("depth tensor",)
|
||||
FUNCTION = "renormalize_depth"
|
||||
CATEGORY = "Camera/depth"
|
||||
CATEGORY = "Camera/Depth"
|
||||
|
||||
def renormalize_depth(
|
||||
self,
|
||||
|
||||
@@ -0,0 +1,287 @@
|
||||
"""Second July-2026 workflow pass: apply the improvements proposed in
|
||||
docs/workflows_review.md.
|
||||
|
||||
1. Point every LoadImage at the bundled example inputs (example_inputs/ is
|
||||
copied into ComfyUI's input dir by install.py), so workflows run on a
|
||||
fresh install without hunting for files.
|
||||
2. video_camera.json dependency trim: drop the dead-end BlurMaskFast
|
||||
(Image-Filters), the easy-mathInt pad computation (Easy-Use) and the now
|
||||
dangling VHS_VideoInfo; swap KJNodes' GetImageRangeFromBatch for the
|
||||
built-in ImageFromBatch. Remaining pack deps: VHS + Florence2.
|
||||
3. Create workflows/record_trajectory.json (TransformToMatrix x2 ->
|
||||
CameraInterpolationNode -> SaveTrajectory) - the missing producer for the
|
||||
trajectory files PC_enricher / wan_vace_ref_to_video consume.
|
||||
4. Stamp `extra.camera_comfyui_rev = 1` in every workflow for future
|
||||
migrations.
|
||||
|
||||
Run: python notebooks/apply_improvements_2026_07.py
|
||||
"""
|
||||
|
||||
import glob
|
||||
import json
|
||||
import os
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
WF = os.path.join(REPO, "workflows")
|
||||
|
||||
PINHOLE = "camera_example_pinhole.jpg"
|
||||
FISHEYE = "camera_example_fisheye.jpg"
|
||||
|
||||
# workflow file -> example image for its LoadImage node(s)
|
||||
LOADIMAGE_DEFAULTS = {
|
||||
"demo_camera_workflow.json": PINHOLE,
|
||||
"Outpaint_node_test.json": PINHOLE,
|
||||
"Outpaint_fisheye180.json": PINHOLE,
|
||||
"outpainting_fisheye_flux.json": PINHOLE,
|
||||
"legacy/outpainting_fisheye.json": PINHOLE,
|
||||
"PointCloud.json": PINHOLE,
|
||||
"pointcloud_walker.json": PINHOLE,
|
||||
"pointcloud_inpaint.json": PINHOLE,
|
||||
"fisheye_to_pointcloud.json": FISHEYE,
|
||||
"legacy/Fisheye_depth_workflow.json": FISHEYE,
|
||||
"PC_enricher.json": FISHEYE,
|
||||
"sbs180_workflow.json": FISHEYE,
|
||||
"wan_vace_ref_to_video.json": FISHEYE,
|
||||
}
|
||||
|
||||
|
||||
def load(rel):
|
||||
with open(os.path.join(WF, rel), encoding="utf-8") as fh:
|
||||
return json.load(fh)
|
||||
|
||||
|
||||
def save(rel, data):
|
||||
with open(os.path.join(WF, rel), "w", encoding="utf-8", newline="\n") as fh:
|
||||
json.dump(data, fh, indent=2, ensure_ascii=False)
|
||||
fh.write("\n")
|
||||
|
||||
|
||||
def node(data, nid):
|
||||
for n in data["nodes"]:
|
||||
if n["id"] == nid:
|
||||
return n
|
||||
raise KeyError(nid)
|
||||
|
||||
|
||||
def set_example_inputs():
|
||||
for rel, image in LOADIMAGE_DEFAULTS.items():
|
||||
d = load(rel)
|
||||
changed = 0
|
||||
for n in d["nodes"]:
|
||||
if n["type"] == "LoadImage":
|
||||
n["widgets_values"][0] = image
|
||||
changed += 1
|
||||
assert changed, f"{rel}: no LoadImage found"
|
||||
save(rel, d)
|
||||
print(f"[ok] {rel}: {changed} LoadImage -> {image}")
|
||||
|
||||
|
||||
def remove_node(data, nid):
|
||||
"""Remove a node and every link touching it; re-index dst slots."""
|
||||
n = node(data, nid)
|
||||
dead = set()
|
||||
for inp in n.get("inputs") or []:
|
||||
if inp.get("link") is not None:
|
||||
dead.add(inp["link"])
|
||||
for out in n.get("outputs") or []:
|
||||
dead.update(out.get("links") or [])
|
||||
data["nodes"] = [x for x in data["nodes"] if x["id"] != nid]
|
||||
data["links"] = [l for l in data["links"] if l[0] not in dead]
|
||||
for x in data["nodes"]:
|
||||
for inp in x.get("inputs") or []:
|
||||
if inp.get("link") in dead:
|
||||
inp["link"] = None
|
||||
for out in x.get("outputs") or []:
|
||||
if out.get("links"):
|
||||
out["links"] = [l for l in out["links"] if l not in dead]
|
||||
|
||||
|
||||
def drop_input(data, nid, name):
|
||||
"""Remove a (widget-converted) input socket and fix slot indices."""
|
||||
n = node(data, nid)
|
||||
inputs = n.get("inputs") or []
|
||||
n["inputs"] = [i for i in inputs if i["name"] != name]
|
||||
index = {i["name"]: k for k, i in enumerate(n["inputs"])}
|
||||
for l in data["links"]:
|
||||
if l[3] == nid:
|
||||
# find by link id which input holds it
|
||||
for i in n["inputs"]:
|
||||
if i.get("link") == l[0]:
|
||||
l[4] = index[i["name"]]
|
||||
|
||||
|
||||
def bbox(data, ids, pad_top=60, pad=20):
|
||||
xs, ys, xe, ye = [], [], [], []
|
||||
for nid in ids:
|
||||
n = node(data, nid)
|
||||
p, s = n["pos"], n["size"]
|
||||
xs.append(p[0]); ys.append(p[1])
|
||||
xe.append(p[0] + s[0]); ye.append(p[1] + s[1])
|
||||
x, y = min(xs) - pad, min(ys) - pad_top
|
||||
return [x, y, max(xe) + pad - x, max(ye) + pad - y]
|
||||
|
||||
|
||||
def fix_video_camera(rel="video_camera.json"):
|
||||
d = load(rel)
|
||||
# 1. dead-end mask blur (only Image-Filters usage; radius was 0/0 anyway)
|
||||
remove_node(d, 60)
|
||||
# 2. pad computation chain (Easy-Use mathInt x2 + VHS_VideoInfo feeding it);
|
||||
# ImagePadForOutpaint falls back to its stored widgets (280/280)
|
||||
for nid in (27, 26, 23):
|
||||
remove_node(d, nid)
|
||||
for name in ("top", "bottom"):
|
||||
drop_input(d, 21, name)
|
||||
n21 = node(d, 21)
|
||||
n21["title"] = "Pad to square — set top/bottom to (W−H)/2"
|
||||
# 3. KJNodes GetImageRangeFromBatch -> built-in ImageFromBatch (frame 0)
|
||||
n63 = node(d, 63)
|
||||
n63["type"] = "ImageFromBatch"
|
||||
n63["properties"]["Node name for S&R"] = "ImageFromBatch"
|
||||
n63["inputs"][0]["name"] = "image"
|
||||
n63["widgets_values"] = [0, 1] # batch_index, length
|
||||
n63["title"] = "First frame (for captioning)"
|
||||
# regroup without the removed nodes
|
||||
groups = [
|
||||
("1. Load & pad video", [9, 21, 68]),
|
||||
("2. Metric video depth", [19, 12]),
|
||||
("3. Re-render with new camera", [16, 11, 10, 6]),
|
||||
("4. Masks & composite", [39, 41, 75, 54, 74]),
|
||||
("5. Auto-caption (Florence2)", [63, 62, 61]),
|
||||
("6. WAN VACE re-generation", [29, 30, 37, 32, 33, 34, 28, 31, 36, 35]),
|
||||
("7. Outputs", [4, 13, 14, 38]),
|
||||
]
|
||||
colors = ["#3f789e", "#a1309b", "#8A8", "#b58b2a", "#88A", "#b06634", "#535"]
|
||||
d["groups"] = [{
|
||||
"id": i + 1, "title": t, "bounding": bbox(d, ids),
|
||||
"color": colors[i % len(colors)], "font_size": 24, "flags": {},
|
||||
} for i, (t, ids) in enumerate(groups)]
|
||||
# refresh the note (dependency list changed)
|
||||
for n in d["nodes"]:
|
||||
if n["type"] == "MarkdownNote" and n.get("title") == "About this workflow":
|
||||
n["widgets_values"] = [(
|
||||
"# Re-shoot a video with a new camera move\n\n"
|
||||
"Re-renders an input video along a user-defined camera "
|
||||
"trajectory and uses WAN 2.1 VACE to regenerate what the new "
|
||||
"camera reveals:\n\n"
|
||||
"1. Video is padded square (ImagePadForOutpaint — set "
|
||||
"top/bottom to (width−height)/2 for your video) and "
|
||||
"depth-estimated per frame (Video-Depth-Anything metric).\n"
|
||||
"2. **VideoCameraMotionSequence** lifts each frame to a point "
|
||||
"cloud and re-renders it along the SE(3)-interpolated "
|
||||
"trajectory.\n"
|
||||
"3. Disocclusion masks + re-rendered frames become VACE "
|
||||
"control video/masks; **Florence2** auto-captions the clip as "
|
||||
"the prompt; WAN 2.1 VACE 14B fills the gaps.\n\n"
|
||||
"- **Set:** video path (VHS Load Video Path); the camera move "
|
||||
"(two TransformToMatrix poses); pad amounts; override the "
|
||||
"auto-caption in CLIPTextEncode if desired.\n"
|
||||
"- **Requires (packs):** VideoHelperSuite, ComfyUI-Florence2.\n"
|
||||
"- **Requires (models):** `metric_video_depth_anything_vitl"
|
||||
".pth` (install.sh `depth`), WAN 2.1 VACE 14B + umt5-xxl + "
|
||||
"WAN VAE (install.sh `vae`), Florence-2 (auto-download).\n"
|
||||
"- **Outputs:** re-rendered composite WEBM, depth WEBM, final "
|
||||
"VACE clip."
|
||||
)]
|
||||
save(rel, d)
|
||||
print(f"[ok] {rel}: removed BlurMaskFast/mathInt/VideoInfo, "
|
||||
f"ImageFromBatch swap, regrouped")
|
||||
|
||||
|
||||
def make_record_trajectory(rel="record_trajectory.json"):
|
||||
note = (
|
||||
"# Record a camera trajectory\n\n"
|
||||
"Produces the `.npy` trajectory file consumed by `PC_enricher.json` "
|
||||
"and `wan_vace_ref_to_video.json` (LoadTrajectory): two poses are "
|
||||
"SE(3)-interpolated into a smooth 20-step path and saved by "
|
||||
"**SaveTrajectory** to your ComfyUI **output** directory.\n\n"
|
||||
"- **Set:** the end pose (shift XYZ in scene units — metric if the "
|
||||
"cloud came from metric depth — plus theta = pitch, phi = yaw) and "
|
||||
"`num_steps`.\n"
|
||||
"- **Then:** move the saved file from `output/` to `input/` so "
|
||||
"LoadTrajectory can list it. A bundled example "
|
||||
"(`ComfyUITrajectory_00001.npy`) is already installed by install.py.\n"
|
||||
"- Chain more CameraInterpolationNode segments (or use "
|
||||
"CameraTrajectoryNode on a point cloud) for multi-keyframe paths."
|
||||
)
|
||||
d = {
|
||||
"id": "00000000-0000-0000-0000-000000000000",
|
||||
"revision": 0,
|
||||
"last_node_id": 5,
|
||||
"last_link_id": 3,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 1, "type": "TransformToMatrix", "title": "Start pose (identity)",
|
||||
"pos": [-500, 320], "size": [315, 154], "flags": {}, "order": 0,
|
||||
"mode": 0, "inputs": [],
|
||||
"outputs": [{"name": "transformation matrix", "type": "MAT_4X4",
|
||||
"links": [1], "slot_index": 0}],
|
||||
"properties": {"Node name for S&R": "TransformToMatrix"},
|
||||
"widgets_values": [0.0, 0.0, 0.0, 0.0, 0.0],
|
||||
},
|
||||
{
|
||||
"id": 2, "type": "TransformToMatrix", "title": "End pose (edit me)",
|
||||
"pos": [-500, 540], "size": [315, 154], "flags": {}, "order": 1,
|
||||
"mode": 0, "inputs": [],
|
||||
"outputs": [{"name": "transformation matrix", "type": "MAT_4X4",
|
||||
"links": [2], "slot_index": 0}],
|
||||
"properties": {"Node name for S&R": "TransformToMatrix"},
|
||||
"widgets_values": [0.0, 0.0, 0.3, 0.0, 30.0],
|
||||
},
|
||||
{
|
||||
"id": 3, "type": "CameraInterpolationNode",
|
||||
"title": "Interpolate 20 poses",
|
||||
"pos": [-120, 430], "size": [226, 78], "flags": {}, "order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{"name": "initial_matrix", "type": "MAT_4X4", "link": 1},
|
||||
{"name": "final_matrix", "type": "MAT_4X4", "link": 2},
|
||||
],
|
||||
"outputs": [{"name": "trajectory", "type": "TENSOR",
|
||||
"links": [3], "slot_index": 0}],
|
||||
"properties": {"Node name for S&R": "CameraInterpolationNode"},
|
||||
"widgets_values": [20],
|
||||
},
|
||||
{
|
||||
"id": 4, "type": "SaveTrajectory", "title": "Save to output/*.npy",
|
||||
"pos": [170, 430], "size": [315, 82], "flags": {}, "order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [{"name": "trajectory", "type": "TENSOR", "link": 3}],
|
||||
"outputs": [],
|
||||
"properties": {"Node name for S&R": "SaveTrajectory"},
|
||||
"widgets_values": ["ComfyUITrajectory"],
|
||||
},
|
||||
{
|
||||
"id": 5, "type": "MarkdownNote", "title": "About this workflow",
|
||||
"pos": [-1080, 320], "size": [520, 430], "flags": {}, "order": 4,
|
||||
"mode": 0, "inputs": [], "outputs": [], "properties": {},
|
||||
"widgets_values": [note], "color": "#432", "bgcolor": "#653",
|
||||
},
|
||||
],
|
||||
"links": [
|
||||
[1, 1, 0, 3, 0, "MAT_4X4"],
|
||||
[2, 2, 0, 3, 1, "MAT_4X4"],
|
||||
[3, 3, 0, 4, 0, "TENSOR"],
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {},
|
||||
"version": 0.4,
|
||||
}
|
||||
save(rel, d)
|
||||
print(f"[ok] created {rel}")
|
||||
|
||||
|
||||
def stamp_revision():
|
||||
for path in glob.glob(os.path.join(WF, "**", "*.json"), recursive=True):
|
||||
rel = os.path.relpath(path, WF)
|
||||
d = load(rel)
|
||||
d.setdefault("extra", {})["camera_comfyui_rev"] = 1
|
||||
save(rel, d)
|
||||
print("[ok] stamped extra.camera_comfyui_rev = 1")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
set_example_inputs()
|
||||
fix_video_camera()
|
||||
make_record_trajectory()
|
||||
stamp_revision()
|
||||
@@ -0,0 +1,294 @@
|
||||
"""Build workflows/fisheye_static_video_to_4d.json.
|
||||
|
||||
Static-fisheye-camera variant of video_to_4d_world.json: because the camera
|
||||
does not move, no VGGT pose estimation is needed — the trajectory is identity,
|
||||
per-frame depth comes from the batched FisheyeDepthEstimator, and the whole
|
||||
static background is a single FisheyeToGaussian prediction of frame 0 split by
|
||||
the motion mask (outside = static world, inside = dynamic canonical).
|
||||
|
||||
Node facts this graph relies on (verified against current INPUT_TYPES):
|
||||
- FisheyeDepthEstimator is batched: IMAGE [T,H,W,C] -> depthmap [T,H,W,1];
|
||||
GS4D nodes squeeze the trailing channel (GS4D_nodes.py::167).
|
||||
- MotionMaskFromDepth interpolates a [K,4,4] trajectory to T internally.
|
||||
- TracksToTrajectories' trajectory input is optional and defaults to identity
|
||||
("static camera" per its tooltip) — left unconnected on purpose.
|
||||
- SplitSplatsByMask returns (inside_splats, outside_splats); its optional
|
||||
camera_matrix defaults to identity, which is exactly the frame-0 camera here.
|
||||
|
||||
Run: python notebooks/build_fisheye_static_4d.py
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
OUT = os.path.join(REPO, "workflows", "fisheye_static_video_to_4d.json")
|
||||
|
||||
NODES = []
|
||||
LINKS = []
|
||||
_link_id = 0
|
||||
|
||||
|
||||
def node(nid, ntype, title, pos, size, widgets, inputs=(), outputs=(), mode=0,
|
||||
extra=None):
|
||||
n = {
|
||||
"id": nid, "type": ntype, "pos": list(pos), "size": list(size),
|
||||
"flags": {}, "order": len(NODES), "mode": mode,
|
||||
"inputs": [dict(i) for i in inputs],
|
||||
"outputs": [
|
||||
{"name": o[0], "type": o[1], "links": [], "slot_index": k}
|
||||
for k, o in enumerate(outputs)
|
||||
],
|
||||
"properties": {"Node name for S&R": ntype},
|
||||
"widgets_values": widgets,
|
||||
}
|
||||
if title:
|
||||
n["title"] = title
|
||||
if extra:
|
||||
n.update(extra)
|
||||
NODES.append(n)
|
||||
return n
|
||||
|
||||
|
||||
def link(src, src_slot, dst, dst_input, ltype):
|
||||
global _link_id
|
||||
_link_id += 1
|
||||
sn = next(n for n in NODES if n["id"] == src)
|
||||
dn = next(n for n in NODES if n["id"] == dst)
|
||||
slot = next(i for i, inp in enumerate(dn["inputs"]) if inp["name"] == dst_input)
|
||||
dn["inputs"][slot]["link"] = _link_id
|
||||
sn["outputs"][src_slot]["links"].append(_link_id)
|
||||
LINKS.append([_link_id, src, src_slot, dst, slot, ltype])
|
||||
|
||||
|
||||
def inp(name, ltype):
|
||||
return {"name": name, "type": ltype, "link": None}
|
||||
|
||||
|
||||
NOTE = """# Static fisheye video → 4D Gaussian video
|
||||
|
||||
Turns a video shot on a **static 180° fisheye camera** (locked-off shot,
|
||||
security cam, tripod) into a 4D (3D + time) Gaussian scene, then re-renders it
|
||||
from a *moving* novel camera.
|
||||
|
||||
A static camera needs **no pose estimation** (no VGGT, unlike
|
||||
`video_to_4d_world.json`): the trajectory is identity, and one frame already
|
||||
sees the entire static background.
|
||||
|
||||
**Stages**
|
||||
1. **FisheyeDepthEstimator** — per-frame metric *radial* depth on the whole
|
||||
fisheye batch (DISTANCE_AWARE multi-view merge).
|
||||
2. **MotionMaskFromDepth** (identity trajectory) — pixels whose depth changes
|
||||
over time are dynamic.
|
||||
3. **FisheyeToGaussian** on frame 0 → whole-scene splats;
|
||||
**SplitSplatsByMask** (FISHEYE 180°) separates the dynamic canonical
|
||||
(inside) from the static world (outside).
|
||||
4. **EstimateTracks** (CoTracker3) + **TracksToTrajectories** (FISHEYE 180°,
|
||||
identity poses — its default) lift 2D tracks to 3D control trajectories.
|
||||
5. **BuildSplats4D** binds the canonical splats to the tracks → 4D scene with
|
||||
the static background attached.
|
||||
6. **RenderSplats4DVideo** renders a novel orbit (pinhole 90°). A second,
|
||||
**muted** render replays the original static fisheye view for A/B
|
||||
comparison — unmute it (Ctrl+M) to use.
|
||||
|
||||
**Set:** the video path + `frame_load_cap` (≤ 64 recommended); the novel-path
|
||||
end pose; `threshold` in MotionMaskFromDepth (raise it if the mask flickers —
|
||||
per-frame depth is not temporally consistent).
|
||||
|
||||
**Requires:** SHARP + gsplat (auto via install.py), CoTracker3 (torch.hub,
|
||||
first-use download), Depth-Anything V2 (auto), VideoHelperSuite pack.
|
||||
|
||||
**Limitations:** the static world only knows what frame 0 saw — regions
|
||||
occluded by moving objects at t=0 are holes if the novel camera peeks behind
|
||||
them; strong depth flicker can leak static pixels into the dynamic set.
|
||||
|
||||
**Outputs:** novel-path WEBM, the 4D scene as `.npz` (SaveSplats4D — reload
|
||||
with LoadSplats4D), optional replay WEBM."""
|
||||
|
||||
DA = "Depth-Anything-V2-Metric-Indoor-Base-hf"
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
node(30, "MarkdownNote", "About this workflow", (-2000, 260), (560, 700), [NOTE],
|
||||
extra={"color": "#432", "bgcolor": "#653"})
|
||||
|
||||
node(1, "VHS_LoadVideoPath", "Load fisheye video (static camera, <= 64 frames)",
|
||||
(-1380, 300), (240, 262),
|
||||
{
|
||||
"video": "input/fisheye_video.mp4",
|
||||
"force_rate": 0, "custom_width": 0, "custom_height": 0,
|
||||
"frame_load_cap": 49, "skip_first_frames": 0, "select_every_nth": 1,
|
||||
"format": "AnimateDiff",
|
||||
"videopreview": {"hidden": False, "paused": False, "params": {
|
||||
"filename": "input/fisheye_video.mp4", "type": "path",
|
||||
"format": "video/mp4", "force_rate": 0, "custom_width": 0,
|
||||
"custom_height": 0, "frame_load_cap": 49,
|
||||
"skip_first_frames": 0, "select_every_nth": 1}},
|
||||
},
|
||||
inputs=[inp("meta_batch", "VHS_BatchManager"), inp("vae", "VAE")],
|
||||
outputs=[("IMAGE", "IMAGE"), ("frame_count", "INT"),
|
||||
("audio", "AUDIO"), ("video_info", "VHS_VIDEOINFO")])
|
||||
|
||||
node(2, "FisheyeDepthEstimator", "Per-frame radial depth (batched)",
|
||||
(-1060, 300), (315, 246),
|
||||
[DA, 1.0, 90.0, 518, 1024, "DISTANCE_AWARE", 25, 1],
|
||||
inputs=[inp("image", "IMAGE")],
|
||||
outputs=[("depthmap", "TENSOR"), ("mask", "MASK")])
|
||||
|
||||
node(3, "TransformToMatrix", "Static camera (identity pose)",
|
||||
(-1060, 640), (315, 154), [0.0, 0.0, 0.0, 0.0, 0.0],
|
||||
outputs=[("transformation matrix", "MAT_4X4")])
|
||||
|
||||
node(4, "CameraInterpolationNode", "Identity trajectory (interpolated to T)",
|
||||
(-680, 680), (226, 78), [2],
|
||||
inputs=[inp("initial_matrix", "MAT_4X4"), inp("final_matrix", "MAT_4X4")],
|
||||
outputs=[("trajectory", "TENSOR")])
|
||||
|
||||
node(5, "MotionMaskFromDepth", "Dynamic-pixel mask (1 = moving)",
|
||||
(-680, 300), (315, 202), ["FISHEYE", 180.0, 0.15, 4, 2, "auto"],
|
||||
inputs=[inp("depth_seq", "TENSOR"), inp("trajectory", "TENSOR")],
|
||||
outputs=[("motion_mask", "MASK")])
|
||||
|
||||
node(6, "ImageFromBatch", "Frame 0 (canonical view)",
|
||||
(-680, 560), (226, 82), [0, 1],
|
||||
inputs=[inp("image", "IMAGE")],
|
||||
outputs=[("IMAGE", "IMAGE")])
|
||||
|
||||
node(7, "FisheyeToGaussian", "SHARP frame 0 -> whole-scene splats",
|
||||
(-300, 460), (330, 290),
|
||||
[180.0, 0, 0, "<download default>", "auto", 90.0, 0, "smart", 0.01, 5.0],
|
||||
inputs=[inp("image", "IMAGE")],
|
||||
outputs=[("splats", "GSPLAT")])
|
||||
|
||||
node(8, "EstimateTracks", "CoTracker3 (downloads on first use)",
|
||||
(-300, 820), (300, 102), [20, "auto"],
|
||||
inputs=[inp("frames", "IMAGE")],
|
||||
outputs=[("tracks", "TENSOR"), ("visibility", "TENSOR")])
|
||||
|
||||
node(9, "SplitSplatsByMask", "Split: inside = dynamic, outside = static world",
|
||||
(100, 300), (315, 174), ["FISHEYE", 180.0, 0.5, "auto"],
|
||||
inputs=[inp("splats", "GSPLAT"), inp("mask", "MASK"),
|
||||
inp("camera_matrix", "MAT_4X4")],
|
||||
outputs=[("inside_splats", "GSPLAT"), ("outside_splats", "GSPLAT")])
|
||||
|
||||
node(10, "TracksToTrajectories", "Lift tracks to 3D (identity poses = default)",
|
||||
(100, 700), (315, 190), ["FISHEYE", 180.0, 0.5, "auto"],
|
||||
inputs=[inp("tracks", "TENSOR"), inp("visibility", "TENSOR"),
|
||||
inp("depth_seq", "TENSOR"), inp("trajectory", "TENSOR")],
|
||||
outputs=[("trajectories3d", "TENSOR"), ("track_valid", "TENSOR")])
|
||||
|
||||
node(11, "BuildSplats4D", "Bind canonical to tracks -> 4D scene",
|
||||
(500, 440), (315, 190), [0, 4, 0.0, "auto"],
|
||||
inputs=[inp("canonical", "GSPLAT"), inp("trajectories3d", "TENSOR"),
|
||||
inp("static", "GSPLAT"), inp("times", "TENSOR"),
|
||||
inp("track_valid", "TENSOR")],
|
||||
outputs=[("splats4d", "GSPLAT4D")])
|
||||
|
||||
node(12, "TransformToMatrix", "Novel path: start (original camera)",
|
||||
(500, 720), (315, 154), [0.0, 0.0, 0.0, 0.0, 0.0],
|
||||
outputs=[("transformation matrix", "MAT_4X4")])
|
||||
|
||||
node(13, "TransformToMatrix", "Novel path: end (small orbit — edit me)",
|
||||
(500, 920), (315, 154), [0.15, 0.0, 0.1, 0.0, -10.0],
|
||||
outputs=[("transformation matrix", "MAT_4X4")])
|
||||
|
||||
node(14, "CameraInterpolationNode", "Novel camera path",
|
||||
(880, 820), (226, 78), [2],
|
||||
inputs=[inp("initial_matrix", "MAT_4X4"), inp("final_matrix", "MAT_4X4")],
|
||||
outputs=[("trajectory", "TENSOR")])
|
||||
|
||||
node(15, "RenderSplats4DVideo", "Render 4D along novel path (pinhole 90°)",
|
||||
(900, 300), (315, 266), [49, 0.0, 1.0, "PINHOLE", 90.0, 768, 768, "auto", 0, "auto"],
|
||||
inputs=[inp("splats4d", "GSPLAT4D"), inp("trajectory", "TENSOR")],
|
||||
outputs=[("images", "IMAGE"), ("masks", "MASK"), ("disparity", "TENSOR")])
|
||||
|
||||
node(16, "RenderSplats4DVideo", "MUTED: replay original fisheye view (A/B check)",
|
||||
(900, 1060), (315, 266), [49, 0.0, 1.0, "FISHEYE", 180.0, 1024, 1024, "auto", 0, "auto"],
|
||||
inputs=[inp("splats4d", "GSPLAT4D"), inp("trajectory", "TENSOR")],
|
||||
outputs=[("images", "IMAGE"), ("masks", "MASK"), ("disparity", "TENSOR")],
|
||||
mode=2)
|
||||
|
||||
node(17, "SaveWEBM", None, (1300, 300), (315, 437),
|
||||
["4d_fisheye_novel", "vp9", 24, 32],
|
||||
inputs=[inp("images", "IMAGE")])
|
||||
|
||||
node(18, "SaveSplats4D", "Save 4D scene (.npz)", (1300, 800), (315, 106),
|
||||
["ComfyUISplat4D_fisheye", False],
|
||||
inputs=[inp("splats4d", "GSPLAT4D")])
|
||||
|
||||
node(19, "SaveWEBM", "MUTED: replay output", (1300, 1060), (315, 437),
|
||||
["4d_fisheye_replay", "vp9", 24, 32],
|
||||
inputs=[inp("images", "IMAGE")], mode=2)
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
link(1, 0, 2, "image", "IMAGE")
|
||||
link(1, 0, 6, "image", "IMAGE")
|
||||
link(1, 0, 8, "frames", "IMAGE")
|
||||
link(2, 0, 5, "depth_seq", "TENSOR")
|
||||
link(2, 0, 10, "depth_seq", "TENSOR")
|
||||
link(3, 0, 4, "initial_matrix", "MAT_4X4")
|
||||
link(3, 0, 4, "final_matrix", "MAT_4X4")
|
||||
link(4, 0, 5, "trajectory", "TENSOR")
|
||||
link(4, 0, 16, "trajectory", "TENSOR")
|
||||
link(5, 0, 9, "mask", "MASK")
|
||||
link(6, 0, 7, "image", "IMAGE")
|
||||
link(7, 0, 9, "splats", "GSPLAT")
|
||||
link(8, 0, 10, "tracks", "TENSOR")
|
||||
link(8, 1, 10, "visibility", "TENSOR")
|
||||
link(9, 0, 11, "canonical", "GSPLAT")
|
||||
link(9, 1, 11, "static", "GSPLAT")
|
||||
link(10, 0, 11, "trajectories3d", "TENSOR")
|
||||
link(10, 1, 11, "track_valid", "TENSOR")
|
||||
link(11, 0, 15, "splats4d", "GSPLAT4D")
|
||||
link(11, 0, 16, "splats4d", "GSPLAT4D")
|
||||
link(11, 0, 18, "splats4d", "GSPLAT4D")
|
||||
link(12, 0, 14, "initial_matrix", "MAT_4X4")
|
||||
link(13, 0, 14, "final_matrix", "MAT_4X4")
|
||||
link(14, 0, 15, "trajectory", "TENSOR")
|
||||
link(15, 0, 17, "images", "IMAGE")
|
||||
link(16, 0, 19, "images", "IMAGE")
|
||||
# NOTE: TracksToTrajectories.trajectory stays unconnected on purpose — its
|
||||
# default is identity poses, i.e. exactly the static camera.
|
||||
|
||||
|
||||
def bbox(ids, pad_top=60, pad=20):
|
||||
ns = [n for n in NODES if n["id"] in ids]
|
||||
x = min(n["pos"][0] for n in ns) - pad
|
||||
y = min(n["pos"][1] for n in ns) - pad_top
|
||||
x2 = max(n["pos"][0] + n["size"][0] for n in ns) + pad
|
||||
y2 = max(n["pos"][1] + n["size"][1] for n in ns) + pad
|
||||
return [x, y, x2 - x, y2 - y]
|
||||
|
||||
|
||||
GROUPS = [
|
||||
("1. Load fisheye video", [1]),
|
||||
("2. Per-frame fisheye depth", [2]),
|
||||
("3. Static camera trajectory", [3, 4]),
|
||||
("4. Motion mask", [5]),
|
||||
("5. Frame-0 splats & static/dynamic split", [6, 7, 9]),
|
||||
("6. Dynamic tracks -> 3D", [8, 10]),
|
||||
("7. 4D scene", [11, 18]),
|
||||
("8. Render novel path (+ muted replay)", [12, 13, 14, 15, 16, 17, 19]),
|
||||
]
|
||||
COLORS = ["#3f789e", "#a1309b", "#8A8", "#b58b2a", "#88A", "#b06634", "#535",
|
||||
"#3f789e"]
|
||||
|
||||
workflow = {
|
||||
"id": "00000000-0000-0000-0000-000000000000",
|
||||
"revision": 0,
|
||||
"last_node_id": 30,
|
||||
"last_link_id": _link_id,
|
||||
"nodes": NODES,
|
||||
"links": LINKS,
|
||||
"groups": [{
|
||||
"id": i + 1, "title": t, "bounding": bbox(ids),
|
||||
"color": COLORS[i % len(COLORS)], "font_size": 24, "flags": {},
|
||||
} for i, (t, ids) in enumerate(GROUPS)],
|
||||
"config": {},
|
||||
"extra": {"camera_comfyui_rev": 1},
|
||||
"version": 0.4,
|
||||
}
|
||||
|
||||
with open(OUT, "w", encoding="utf-8", newline="\n") as fh:
|
||||
json.dump(workflow, fh, indent=2, ensure_ascii=False)
|
||||
fh.write("\n")
|
||||
print(f"wrote {OUT}: {len(NODES)} nodes, {len(LINKS)} links")
|
||||
@@ -0,0 +1,699 @@
|
||||
"""One-shot July 2026 workflow rework: schema repairs + documentation.
|
||||
|
||||
For every workflows/*.json this script:
|
||||
1. migrates widgets_values stored with pre-2026 node schemas to the current
|
||||
widget lists (see notebooks/validate_workflows.py for the detector);
|
||||
2. fixes broken references (wan UNET filename typo, ReprojectImage rotation
|
||||
widgets that no longer exist -> explicit TransformToMatrix nodes);
|
||||
3. adds an embedded MarkdownNote explaining purpose/stages/requirements,
|
||||
meaningful group boxes and node titles;
|
||||
4. rewrites the JSON pretty-printed (indent 2).
|
||||
|
||||
Idempotent-ish: repairs are guarded by length/value checks, notes are only
|
||||
added if no MarkdownNote titled 'About this workflow' exists.
|
||||
|
||||
Run: python notebooks/rework_workflows_2026_07.py
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
WF = os.path.join(REPO, "workflows")
|
||||
|
||||
DA_MODEL = "Depth-Anything-V2-Metric-Indoor-Base-hf"
|
||||
|
||||
NOTE_TITLE = "About this workflow"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Generic helpers
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def load(name):
|
||||
with open(os.path.join(WF, name), encoding="utf-8") as fh:
|
||||
return json.load(fh)
|
||||
|
||||
|
||||
def save(name, data):
|
||||
# normalize pos/size dict-form (pre-2024 serialization) to arrays
|
||||
for n in data.get("nodes", []):
|
||||
for key in ("pos", "size"):
|
||||
v = n.get(key)
|
||||
if isinstance(v, dict):
|
||||
n[key] = [v[k] for k in sorted(v, key=lambda s: int(s))]
|
||||
with open(os.path.join(WF, name), "w", encoding="utf-8", newline="\n") as fh:
|
||||
json.dump(data, fh, indent=2, ensure_ascii=False)
|
||||
fh.write("\n")
|
||||
|
||||
|
||||
def node(data, nid):
|
||||
for n in data["nodes"]:
|
||||
if n["id"] == nid:
|
||||
return n
|
||||
raise KeyError(f"node {nid} not found")
|
||||
|
||||
|
||||
def xy(v):
|
||||
if isinstance(v, dict):
|
||||
return [float(v["0"]), float(v["1"])]
|
||||
return [float(v[0]), float(v[1])]
|
||||
|
||||
|
||||
def bbox(data, ids, pad_top=60, pad=20):
|
||||
xs, ys, xe, ye = [], [], [], []
|
||||
for nid in ids:
|
||||
n = node(data, nid)
|
||||
p, s = xy(n["pos"]), xy(n["size"])
|
||||
xs.append(p[0]); ys.append(p[1])
|
||||
xe.append(p[0] + s[0]); ye.append(p[1] + s[1])
|
||||
x, y = min(xs) - pad, min(ys) - pad_top
|
||||
return [x, y, max(xe) + pad - x, max(ye) + pad - y]
|
||||
|
||||
|
||||
GROUP_COLORS = ["#3f789e", "#a1309b", "#8A8", "#b58b2a", "#88A", "#b06634", "#535"]
|
||||
|
||||
|
||||
def set_groups(data, groups):
|
||||
"""groups: list of (title, node_ids). Replaces the groups list."""
|
||||
out = []
|
||||
for i, (title, ids) in enumerate(groups):
|
||||
out.append({
|
||||
"id": i + 1,
|
||||
"title": title,
|
||||
"bounding": bbox(data, ids),
|
||||
"color": GROUP_COLORS[i % len(GROUP_COLORS)],
|
||||
"font_size": 24,
|
||||
"flags": {},
|
||||
})
|
||||
data["groups"] = out
|
||||
|
||||
|
||||
def retitle_groups(data, titles):
|
||||
"""titles: list matching data['groups'] order."""
|
||||
assert len(titles) == len(data.get("groups", [])), "group count mismatch"
|
||||
for g, t in zip(data["groups"], titles):
|
||||
g["title"] = t
|
||||
|
||||
|
||||
def set_title(data, nid, title):
|
||||
node(data, nid)["title"] = title
|
||||
|
||||
|
||||
def next_node_id(data):
|
||||
data["last_node_id"] = int(data.get("last_node_id", 0)) + 1
|
||||
return data["last_node_id"]
|
||||
|
||||
|
||||
def next_link_id(data):
|
||||
data["last_link_id"] = int(data.get("last_link_id", 0)) + 1
|
||||
return data["last_link_id"]
|
||||
|
||||
|
||||
def add_note(data, markdown, pos=None, size=None):
|
||||
for n in data["nodes"]:
|
||||
if n["type"] in ("MarkdownNote", "Note") and n.get("title") == NOTE_TITLE:
|
||||
n["widgets_values"] = [markdown]
|
||||
return n["id"]
|
||||
if pos is None:
|
||||
xs = [xy(n["pos"])[0] for n in data["nodes"]]
|
||||
ys = [xy(n["pos"])[1] for n in data["nodes"]]
|
||||
pos = [min(xs) - 560, min(ys)]
|
||||
if size is None:
|
||||
lines = markdown.count("\n") + 1
|
||||
size = [520, max(220, min(700, 26 * lines + 80))]
|
||||
nid = next_node_id(data)
|
||||
data["nodes"].append({
|
||||
"id": nid,
|
||||
"type": "MarkdownNote",
|
||||
"title": NOTE_TITLE,
|
||||
"pos": pos,
|
||||
"size": size,
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [markdown],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653",
|
||||
})
|
||||
return nid
|
||||
|
||||
|
||||
def add_transform_node(data, pos, widgets):
|
||||
nid = next_node_id(data)
|
||||
data["nodes"].append({
|
||||
"id": nid,
|
||||
"type": "TransformToMatrix",
|
||||
"pos": pos,
|
||||
"size": [315, 154],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [{"name": "transformation matrix", "type": "MAT_4X4",
|
||||
"links": [], "slot_index": 0}],
|
||||
"properties": {"Node name for S&R": "TransformToMatrix"},
|
||||
"widgets_values": widgets,
|
||||
})
|
||||
return nid
|
||||
|
||||
|
||||
def link(data, src, dst, dst_input, ltype="MAT_4X4", src_slot=0):
|
||||
"""Connect src node output slot to dst node's named input (added if absent)."""
|
||||
dnode = node(data, dst)
|
||||
inputs = dnode.setdefault("inputs", [])
|
||||
slot = None
|
||||
for i, inp in enumerate(inputs):
|
||||
if inp["name"] == dst_input:
|
||||
slot = i
|
||||
break
|
||||
if slot is None:
|
||||
inputs.append({"name": dst_input, "type": ltype, "link": None})
|
||||
slot = len(inputs) - 1
|
||||
lid = next_link_id(data)
|
||||
inputs[slot]["link"] = lid
|
||||
snode = node(data, src)
|
||||
snode["outputs"][src_slot].setdefault("links", None)
|
||||
if snode["outputs"][src_slot]["links"] is None:
|
||||
snode["outputs"][src_slot]["links"] = []
|
||||
snode["outputs"][src_slot]["links"].append(lid)
|
||||
data["links"].append([lid, src, src_slot, dst, slot, ltype])
|
||||
return lid
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Global widget-schema repairs (2025 -> 2026 node schemas)
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def repair_widgets(data):
|
||||
changed = []
|
||||
for n in data["nodes"]:
|
||||
t, w = n["type"], n.get("widgets_values")
|
||||
if not isinstance(w, list):
|
||||
continue
|
||||
if t == "DepthEstimatorNode" and len(w) == 2:
|
||||
n["widgets_values"] = w + [1] # median_blur_kernel
|
||||
elif t == "FisheyeDepthEstimator" and len(w) == 6:
|
||||
# old: [model, scale, pfov, pres, fres, softmerge_radius]
|
||||
n["widgets_values"] = w[:5] + ["SOFTMERGE", w[5], 1]
|
||||
elif t == "PointCloudCleaner" and len(w) == 2:
|
||||
# old (voxel_size, min_points) were world-unit; semantics changed
|
||||
n["widgets_values"] = [1024, 1024, 1.0, 3]
|
||||
elif t == "CameraMotionNode" and len(w) == 6:
|
||||
n["widgets_values"] = w + [0, False, False]
|
||||
elif t == "CameraInterpolationNode" and len(w) == 0:
|
||||
n["widgets_values"] = [2] # keyframes; renderers interpolate frames
|
||||
elif t == "PointcloudTrajectoryEnricher" and len(w) == 20:
|
||||
# old tail: reproject-back proj/fov/w/h + legacy backend/voxel params
|
||||
n["widgets_values"] = w[:13] + [0.07, 3, DA_MODEL]
|
||||
else:
|
||||
continue
|
||||
changed.append(f"{t}#{n['id']}")
|
||||
return changed
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Per-file specs
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def do_demo_camera(name="demo_camera_workflow.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 1, "Rotate camera (pitch 60°)")
|
||||
set_title(d, 2, "Pinhole 90° → Equirect 180°")
|
||||
set_title(d, 7, "Reprojected image")
|
||||
set_title(d, 4, "Coverage mask (white = hole)")
|
||||
add_note(d, (
|
||||
"# Camera reprojection demo\n\n"
|
||||
"Minimal example of the two core nodes: **TransformToMatrix** rotates the "
|
||||
"virtual camera (theta = 60° pitch) and **ReprojectImage** converts a 90° "
|
||||
"pinhole image into a 180° equirectangular view.\n\n"
|
||||
"Previews show the reprojected image and the coverage mask "
|
||||
"(white = pixels the source image cannot see).\n\n"
|
||||
"- **Set:** the image in LoadImage.\n"
|
||||
"- **Requires:** nothing beyond this pack (no models).\n"
|
||||
"- **Try:** switch `output_projection` to FISHEYE, or raise `feathering` "
|
||||
"to soften the mask edge."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_outpaint_node_test(name="Outpaint_node_test.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 11, "Outpaint one patch (yaw +45°)")
|
||||
set_title(d, 3, "Result")
|
||||
set_title(d, 5, "Holes still to fill")
|
||||
add_note(d, (
|
||||
"# OutpaintAnyProjection smoke test\n\n"
|
||||
"Single-node sanity check: a 90° pinhole image is placed on a 180° fisheye "
|
||||
"canvas and one 90° pinhole patch at yaw +45° is Flux-inpainted "
|
||||
"(10 steps for speed).\n\n"
|
||||
"Outputs: the partially outpainted canvas and the *remaining holes* mask — "
|
||||
"chain more OutpaintAnyProjection nodes at other angles to fill it "
|
||||
"(see `Outpaint_fisheye180.json`).\n\n"
|
||||
"- **Set:** input image; prompt inside the node (empty = unconditional).\n"
|
||||
"- **Requires:** `custom_nodes/inpainting_flux` "
|
||||
"(installed automatically by this pack's install.py) — downloads "
|
||||
"FLUX.1-Fill NF4 weights on first run."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_fisheye_to_pointcloud(name="fisheye_to_pointcloud.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 155, "Metric depth (fisheye-aware)")
|
||||
set_title(d, 160, "Depth → point cloud")
|
||||
set_title(d, 161, "Save .ply / .npy")
|
||||
set_groups(d, [
|
||||
("1. Fisheye metric depth", [154, 155, 156, 157, 158, 159]),
|
||||
("2. Unproject & save", [160, 161]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# Fisheye 180° → point cloud\n\n"
|
||||
"Estimates metric depth directly on a 180° fisheye image — "
|
||||
"**FisheyeDepthEstimator** internally splits it into pinhole views, runs "
|
||||
"Depth-Anything V2 on each and merges the depths back (SOFTMERGE) — then "
|
||||
"unprojects image + depth to a 3D point cloud and saves it.\n\n"
|
||||
"Previews: colorized depth and the validity mask.\n\n"
|
||||
"- **Set:** fisheye input image (e.g. produced by "
|
||||
"`Outpaint_fisheye180.json`); filename in SavePointCloud.\n"
|
||||
"- **Requires:** Depth-Anything V2 (auto-downloads from HuggingFace).\n"
|
||||
"- **Next:** view the cloud with `test_pointcloud_loading.json` or "
|
||||
"synthesize a second eye with `sbs180_workflow.json`."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_pointcloud(name="PointCloud.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 6, "Move camera (dolly −0.1, yaw 20°)")
|
||||
set_title(d, 31, "Render novel view")
|
||||
set_title(d, 47, "Drop stretched/occluded points")
|
||||
set_title(d, 43, "Novel-view depth → image")
|
||||
set_groups(d, [
|
||||
("1. Image → metric ray depth", [1, 38, 45, 42, 40]),
|
||||
("2. Lift to 3D & move camera", [18, 6, 9, 47]),
|
||||
("3. Novel view & previews", [31, 13, 12, 43, 44, 25, 26]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# Single image → point cloud → novel view\n\n"
|
||||
"Lifts one pinhole image to a 3D point cloud via monocular metric depth, "
|
||||
"moves the camera, cleans occlusion artifacts and re-renders from the new "
|
||||
"viewpoint.\n\n"
|
||||
"Stages: DepthEstimator → ZDepthToRayDepth → DepthToPointCloud → "
|
||||
"TransformPointCloud (dolly −0.1, yaw 20°) → PointCloudCleaner → "
|
||||
"ProjectPointCloud. MedianFilterImage smooths the re-render "
|
||||
"(from ComfyUI-Image-Filters, optional).\n\n"
|
||||
"- **Set:** input image; the camera move in TransformToMatrix.\n"
|
||||
"- **Requires:** Depth-Anything V2 (auto-download); ComfyUI-Image-Filters "
|
||||
"only for the median-filter preview.\n"
|
||||
"- **Note:** PointCloudCleaner params were reset to defaults during the "
|
||||
"2026-07 schema migration — retune voxel_size / min_points_per_voxel if "
|
||||
"the render looks too sparse."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_pointcloud_walker(name="Pointcloud walker.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 5, "Camera move (edit me)")
|
||||
set_title(d, 17, "Build trajectory")
|
||||
set_title(d, 14, "Render fly-through")
|
||||
set_groups(d, [
|
||||
("1. Image → point cloud", [1, 3, 20, 2]),
|
||||
("2. Trajectory", [5, 4, 17]),
|
||||
("3. Render & save", [14, 10]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# Point cloud walker (fly-through video)\n\n"
|
||||
"Single image → metric depth → point cloud, then **CameraTrajectoryNode** "
|
||||
"derives a camera path and **CameraMotionNode** renders a fly-through "
|
||||
"saved as WEBM.\n\n"
|
||||
"- **Set:** input image; the move in TransformToMatrix "
|
||||
"(shift XYZ + theta/phi); frames-per-segment (`n_points`) in "
|
||||
"CameraMotionNode.\n"
|
||||
"- **Requires:** Depth-Anything V2 (auto-download).\n"
|
||||
"- **Note:** the pinhole FOV here is 60° — keep DepthToPointCloud and "
|
||||
"CameraMotionNode FOVs consistent with your input."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_test_pointcloud_loading(name="Test pointcloud_loading.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 6, "Load saved cloud (.ply/.npy)")
|
||||
set_title(d, 1, "Start pose (identity)")
|
||||
set_title(d, 5, "End pose (dolly +0.1)")
|
||||
set_title(d, 7, "Render motion")
|
||||
add_note(d, (
|
||||
"# Load a saved point cloud & orbit\n\n"
|
||||
"Reloads a point cloud saved by SavePointCloud and renders a short camera "
|
||||
"move between two poses (identity → 0.1 forward) as a WEBM.\n\n"
|
||||
"- **Set:** the file in LoadPointCloud (dropdown lists the ComfyUI input "
|
||||
"dir — run `PointCloud.json` or `fisheye_to_pointcloud.json` first); the "
|
||||
"two TransformToMatrix poses.\n"
|
||||
"- **Requires:** nothing beyond this pack."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_pc_enricher(name="PC_enricher.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 3, "Enrich cloud along trajectory (Flux outpaint)")
|
||||
set_title(d, 11, "Orbit preview of enriched cloud")
|
||||
set_groups(d, [
|
||||
("1. Fisheye → point cloud", [13, 14, 17, 16, 15]),
|
||||
("2. Enrich along trajectory", [18, 3]),
|
||||
("3. Save & preview", [4, 11, 12]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# Point cloud enricher (outpaint along a trajectory)\n\n"
|
||||
"Fisheye image → metric depth → point cloud, then "
|
||||
"**PointcloudTrajectoryEnricher** walks the loaded camera trajectory: at "
|
||||
"each pose it renders the cloud, Flux-inpaints the disocclusion holes, "
|
||||
"re-estimates depth, aligns it and merges the new points into the cloud. "
|
||||
"The enriched cloud is saved and previewed as an orbit video.\n\n"
|
||||
"This is the one-node version of `pointcloud_inpaint.json`.\n\n"
|
||||
"- **Set:** input image; trajectory .npy (record one with "
|
||||
"SaveTrajectory); prompt inside the enricher.\n"
|
||||
"- **Requires:** inpainting_flux (auto-installed), Depth-Anything V2, "
|
||||
"FLUX.1-Fill NF4 weights.\n"
|
||||
"- **Note:** 2026-07 schema migration — the enricher's render/reproject "
|
||||
"options are now internal; voxel merge params were reset to defaults."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_fisheye_depth(name="Fisheye_depth_workflow.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
retitle_groups(d, [
|
||||
"View 1: center pinhole 90°",
|
||||
"View 2: yaw +45°",
|
||||
"View 3: yaw −45°",
|
||||
"View 4: pitch +45°",
|
||||
"View 5: pitch −45°",
|
||||
"View 6: full-fisheye fallback",
|
||||
])
|
||||
# fusion chain + export were never grouped
|
||||
d["groups"].append({
|
||||
"id": 7,
|
||||
"title": "Fuse views & export point cloud",
|
||||
"bounding": bbox(d, [95, 146, 147, 148, 149, 105, 151, 153, 154]),
|
||||
"color": "#b58b2a",
|
||||
"font_size": 24,
|
||||
"flags": {},
|
||||
})
|
||||
set_title(d, 149, "Final fuse (SRC keeps fused views)")
|
||||
set_title(d, 151, "Fused depth → point cloud")
|
||||
add_note(d, (
|
||||
"# Multi-view fisheye depth — manual reference\n\n"
|
||||
"Estimates consistent metric depth over a 180° fisheye image by "
|
||||
"reprojecting it into five pinhole views (center, yaw ±45°, pitch ±45°), "
|
||||
"running Depth-Anything V2 on each, reprojecting the depths back to the "
|
||||
"fisheye and progressively fusing them (CombineDepthsNode + "
|
||||
"DepthRenormalizer to align scales), plus a full-fisheye pass for the "
|
||||
"rim. The fused depth is unprojected and saved as a point cloud.\n\n"
|
||||
"⚠️ **This whole graph is now one node** — `FisheyeDepthEstimator` does "
|
||||
"the same split-and-merge internally (see `fisheye_to_pointcloud.json`). "
|
||||
"Kept as a transparent, tweakable reference implementation.\n\n"
|
||||
"- **Set:** fisheye input image; SavePointCloud filename.\n"
|
||||
"- **Requires:** Depth-Anything V2 (auto-download)."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_outpaint_fisheye180(name="Outpaint_fisheye180.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
retitle_groups(d, [
|
||||
"2. Chained patch outpaints",
|
||||
"1. Pinhole → fisheye canvas",
|
||||
])
|
||||
set_title(d, 2, "Outpaint yaw +45°")
|
||||
set_title(d, 3, "Outpaint yaw −45°")
|
||||
set_title(d, 4, "Outpaint pitch +45°")
|
||||
set_title(d, 5, "Outpaint pitch −45°")
|
||||
set_title(d, 9, "Full-frame pass (low-res rim fill)")
|
||||
set_title(d, 15, "Upscale rim fill to 4096")
|
||||
set_title(d, 12, "Composite sharp patches over rim")
|
||||
add_note(d, (
|
||||
"# Outpaint to a full 180° fisheye\n\n"
|
||||
"Places a 90° pinhole image onto a 180° fisheye canvas (ReprojectImage), "
|
||||
"then chains **five OutpaintAnyProjection passes** — yaw +45°, yaw −45°, "
|
||||
"pitch +45°, pitch −45°, and a low-res full-frame pass for the rim. Each "
|
||||
"pass consumes the previous pass's *remaining holes* mask. A PorterDuff "
|
||||
"composite keeps the sharp high-res patches on top of the upscaled rim "
|
||||
"fill.\n\n"
|
||||
"- **Set:** input image; prompts inside each outpaint node (optional).\n"
|
||||
"- **Requires:** inpainting_flux (auto-installed by install.py); "
|
||||
"FLUX.1-Fill NF4 weights download on first run (~12 GB VRAM).\n"
|
||||
"- **Output:** `Saved_fisheye` PNG — used as the input of "
|
||||
"`fisheye_to_pointcloud.json`, `sbs180_workflow.json` and "
|
||||
"`PC_enricher.json`."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_outpainting_fisheye_sd(name="outpainting_fisheye.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
# --- structural repair: pre-2025 ReprojectImage stored patch rotations as
|
||||
# widgets; the current node takes a MAT_4X4 transform_matrix input and an
|
||||
# inverse flag instead (mirrors outpainting_fisheye_flux.json).
|
||||
fixes = {
|
||||
42: ([90, 180, "PINHOLE", "FISHEYE", 4096, 4096, False, 0], None),
|
||||
40: ([180, 90, "FISHEYE", "PINHOLE", 1024, 1024, False, 0], "+42"),
|
||||
51: ([90, 180, "PINHOLE", "FISHEYE", 4096, 4096, True, 0], "+42"),
|
||||
59: ([180, 90, "FISHEYE", "PINHOLE", 1024, 1024, False, 0], "-42"),
|
||||
63: ([90, 180, "PINHOLE", "FISHEYE", 4096, 4096, True, 0], "-42"),
|
||||
67: ([180, 180, "FISHEYE", "FISHEYE", 1024, 1024, False, 0], None),
|
||||
72: ([180, 180, "FISHEYE", "FISHEYE", 4096, 4096, False, 0], None),
|
||||
}
|
||||
needs_repair = any(len(node(d, nid).get("widgets_values", [])) == 9
|
||||
for nid in fixes)
|
||||
if needs_repair:
|
||||
m_pos = add_transform_node(d, [20, 480], [0, 0, 0, 0, 42])
|
||||
m_neg = add_transform_node(d, [20, 1180], [0, 0, 0, 0, -42])
|
||||
set_title(d, m_pos, "Patch rotation +42°")
|
||||
set_title(d, m_neg, "Patch rotation −42°")
|
||||
for nid, (widgets, mat) in fixes.items():
|
||||
node(d, nid)["widgets_values"] = widgets
|
||||
if mat is not None:
|
||||
link(d, m_pos if mat == "+42" else m_neg, nid, "transform_matrix")
|
||||
retitle_groups(d, [
|
||||
"Patch 1: yaw +42° — extract & inpaint-encode",
|
||||
"Patch 2: yaw −42° — extract & inpaint-encode",
|
||||
])
|
||||
set_title(d, 42, "Pinhole 90° → fisheye canvas")
|
||||
set_title(d, 67, "Full-frame pass (low-res)")
|
||||
add_note(d, (
|
||||
"# Outpaint fisheye — SD-inpaint-checkpoint variant\n\n"
|
||||
"Same pipeline as `outpainting_fisheye_flux.json`, but the holes are "
|
||||
"filled with a classic SD inpainting checkpoint (VAEEncodeForInpaint + "
|
||||
"KSampler) instead of Flux: pinhole 90° → 180° fisheye canvas, two ±42° "
|
||||
"pinhole patches and a final full-frame pass are inpainted and "
|
||||
"composited; RealESRGAN upscales the result.\n\n"
|
||||
"- **Set:** input image; positive/negative prompts.\n"
|
||||
"- **Requires (models):** `512-inpainting-ema.safetensors`, "
|
||||
"`RealESRGAN_x4plus.pth`. No custom node packs.\n"
|
||||
"- **Repaired 2026-07:** patch rotations were stored in a pre-2025 "
|
||||
"ReprojectImage schema; they are now explicit TransformToMatrix (±42°) "
|
||||
"nodes + `inverse` flags. Prefer the Flux variant for quality."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_outpainting_fisheye_flux(name="outpainting_fisheye_flux.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
for nid, inverse in ((42, False), (40, False), (51, True), (80, False),
|
||||
(63, True), (67, False), (72, False)):
|
||||
w = node(d, nid)["widgets_values"]
|
||||
if len(w) == 8:
|
||||
w[6] = inverse # was stored as 0/45/true mixtures
|
||||
retitle_groups(d, [
|
||||
"Patch 1: yaw +42° — extract & Flux inpaint",
|
||||
"Patch 2: yaw −42° — extract & Flux inpaint",
|
||||
])
|
||||
set_title(d, 42, "Pinhole 90° → fisheye canvas")
|
||||
set_title(d, 75, "Patch rotation +42°")
|
||||
set_title(d, 76, "Patch rotation −42°")
|
||||
set_title(d, 67, "Full-frame pass (low-res)")
|
||||
set_title(d, 77, "Upscale full pass to 4096")
|
||||
add_note(d, (
|
||||
"# Outpaint fisheye — Flux variant\n\n"
|
||||
"Pinhole 90° image → 180° fisheye canvas; **Flux Inpainting** fills two "
|
||||
"90° pinhole patches (rotations ±42° from the TransformToMatrix nodes) "
|
||||
"and one low-res full-frame fisheye pass; PorterDuff composites + "
|
||||
"RealESRGAN upscale assemble the final 4096² fisheye "
|
||||
"(saved as `fluxfish`).\n\n"
|
||||
"This is the manual, step-visible version of what "
|
||||
"**OutpaintAnyProjection** does in one node — see "
|
||||
"`Outpaint_fisheye180.json`.\n\n"
|
||||
"- **Set:** input image; prompts in the three Flux Inpainting nodes.\n"
|
||||
"- **Requires:** inpainting_flux (auto-installed), FLUX.1-Fill NF4 "
|
||||
"weights (first-run download), `RealESRGAN_x4plus.pth`."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_sbs180(name="sbs180_workflow.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
retitle_groups(d, [
|
||||
"1. Fisheye metric depth",
|
||||
"2. Point cloud & eye-baseline shift",
|
||||
"3. Outpaint disocclusions & export equirect",
|
||||
"Previews",
|
||||
])
|
||||
set_title(d, 33, "Eye baseline (shiftX 0.1)")
|
||||
set_title(d, 37, "Right eye (equirect)")
|
||||
set_title(d, 44, "Left eye (equirect)")
|
||||
add_note(d, (
|
||||
"# SBS VR180: synthesize the second eye\n\n"
|
||||
"From one 180° fisheye view, synthesizes a stereo pair: fisheye metric "
|
||||
"depth → point cloud → clean → shift the camera by the eye baseline "
|
||||
"(TransformToMatrix shiftX = 0.1) → re-project to fisheye → four chained "
|
||||
"OutpaintAnyProjection passes fill the disocclusions → both eyes are "
|
||||
"exported as 180° equirectangular images.\n\n"
|
||||
"- **Set:** fisheye input (e.g. from `Outpaint_fisheye180.json`); the "
|
||||
"baseline (0.1 ≈ 6.5 cm when depth is metric); prompts optional.\n"
|
||||
"- **Requires:** inpainting_flux (auto-installed), Depth-Anything V2.\n"
|
||||
"- **Output:** `init_camera_equirect` + `shifted_camera_equirect` — "
|
||||
"combine side-by-side for a VR180 player."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_pointcloud_inpaint(name="pointcloud_inpaint.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 6, "Camera shift (dolly −0.1)")
|
||||
set_title(d, 43, "Flux inpaint holes")
|
||||
set_title(d, 54, "Align new depth to cloud")
|
||||
set_title(d, 55, "Merge old + new points")
|
||||
set_groups(d, [
|
||||
("1. Image → point cloud", [1, 46, 18]),
|
||||
("2. Novel view & hole mask", [6, 9, 10, 45, 27, 35, 48, 47]),
|
||||
("3. Flux inpaint", [43, 26]),
|
||||
("4. Lift inpainted region & merge", [49, 54, 50, 55]),
|
||||
("5. Orbit render", [61, 67, 68, 69, 63]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# Iterative point-cloud inpainting\n\n"
|
||||
"Enriches a single-image point cloud with generated content: move the "
|
||||
"camera back → render the cloud (holes appear) → grow + invert the "
|
||||
"coverage mask → **Flux-inpaint the holes** → re-estimate depth on the "
|
||||
"inpainted image → **DepthRenormalizer** aligns it to the original "
|
||||
"cloud's depth → lift the new pixels to 3D → **PointCloudUnion** merges "
|
||||
"everything → orbit render of the enriched scene.\n\n"
|
||||
"One-node alternative: PointcloudTrajectoryEnricher "
|
||||
"(`PC_enricher.json`).\n\n"
|
||||
"- **Set:** input image; camera shift; inpaint prompt.\n"
|
||||
"- **Requires:** inpainting_flux (auto-installed), Depth-Anything V2."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_video_camera(name="video_camera.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_groups(d, [
|
||||
("1. Load & pad video", [9, 23, 26, 27, 21, 68]),
|
||||
("2. Metric video depth", [19, 12]),
|
||||
("3. Re-render with new camera", [16, 11, 10, 6]),
|
||||
("4. Masks & composite", [39, 41, 60, 75, 54, 74]),
|
||||
("5. Auto-caption (Florence2)", [63, 62, 61]),
|
||||
("6. WAN VACE re-generation", [29, 30, 37, 32, 33, 34, 28, 31, 36, 35]),
|
||||
("7. Outputs", [4, 13, 14, 38]),
|
||||
])
|
||||
set_title(d, 11, "Final camera pose (edit me)")
|
||||
set_title(d, 6, "Re-render along trajectory")
|
||||
set_title(d, 28, "WAN VACE control")
|
||||
add_note(d, (
|
||||
"# Re-shoot a video with a new camera move\n\n"
|
||||
"Re-renders an input video along a user-defined camera trajectory and "
|
||||
"uses WAN 2.1 VACE to regenerate what the new camera reveals:\n\n"
|
||||
"1. Video is padded square and depth-estimated per frame "
|
||||
"(Video-Depth-Anything metric).\n"
|
||||
"2. **VideoCameraMotionSequence** lifts each frame to a point cloud and "
|
||||
"re-renders it along the SE(3)-interpolated trajectory.\n"
|
||||
"3. Disocclusion masks + the re-rendered frames become VACE "
|
||||
"control video/masks; **Florence2** auto-captions the clip as the "
|
||||
"prompt; WAN 2.1 VACE 14B fills the gaps.\n\n"
|
||||
"- **Set:** video path (VHS Load Video Path); the camera move "
|
||||
"(two TransformToMatrix poses); override the auto-caption in "
|
||||
"CLIPTextEncode if desired.\n"
|
||||
"- **Requires (packs):** VideoHelperSuite, ComfyUI-Florence2, KJNodes "
|
||||
"(GetImageRangeFromBatch), ComfyUI-Easy-Use (math), ComfyUI-Image-"
|
||||
"Filters (BlurMaskFast).\n"
|
||||
"- **Requires (models):** `metric_video_depth_anything_vitl.pth` "
|
||||
"(install.sh `depth`), WAN 2.1 VACE 14B + umt5-xxl + WAN VAE "
|
||||
"(install.sh `vae`), Florence-2 (auto-download).\n"
|
||||
"- **Outputs:** re-rendered composite WEBM, depth WEBM, final VACE clip."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_wan_vace_ref(name="wan_vace_ref_to_video.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
# broken model filename: comma + missing underscore
|
||||
n37 = node(d, 37)
|
||||
if n37["widgets_values"][0] == "wan2,1_vace14B_fp16.safetensors":
|
||||
n37["widgets_values"][0] = "wan2.1_vace_14B_fp16.safetensors"
|
||||
set_title(d, 70, "Render control frames + masks")
|
||||
set_title(d, 55, "WAN VACE control")
|
||||
set_groups(d, [
|
||||
("1. Fisheye image → point cloud", [52, 62, 83, 82, 81, 63, 84]),
|
||||
("2. Camera-motion control video", [77, 70, 78, 80]),
|
||||
("3. WAN VACE generation", [37, 54, 38, 6, 7, 39, 55, 3, 56, 8]),
|
||||
("4. Save", [60, 85, 58]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# WAN VACE: still image + camera move → video\n\n"
|
||||
"Turns a single 180° fisheye still into a camera-move video: metric "
|
||||
"depth → point cloud → **CameraMotionNode** renders point-splat frames "
|
||||
"and masks along a loaded trajectory; these become the VACE control "
|
||||
"video/masks with the original image as the reference, and WAN 2.1 VACE "
|
||||
"14B synthesizes the final clip from your prompt.\n\n"
|
||||
"- **Set:** fisheye image; trajectory file (record one with "
|
||||
"SaveTrajectory); the positive prompt.\n"
|
||||
"- **Requires:** WAN 2.1 VACE models (install.sh `vae`), Depth-Anything "
|
||||
"V2, VideoHelperSuite (only for the h264/MP4 export — SaveWEBM works "
|
||||
"without it).\n"
|
||||
"- **Fixed 2026-07:** the UNET filename contained a typo "
|
||||
"(`wan2,1_vace14B` → `wan2.1_vace_14B_fp16.safetensors`)."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def main():
|
||||
for fn in (
|
||||
do_demo_camera, do_outpaint_node_test, do_fisheye_to_pointcloud,
|
||||
do_pointcloud, do_pointcloud_walker, do_test_pointcloud_loading,
|
||||
do_pc_enricher, do_fisheye_depth, do_outpaint_fisheye180,
|
||||
do_outpainting_fisheye_sd, do_outpainting_fisheye_flux, do_sbs180,
|
||||
do_pointcloud_inpaint, do_video_camera, do_wan_vace_ref,
|
||||
):
|
||||
fn()
|
||||
print(f"[ok] {fn.__name__}")
|
||||
# video_to_4d_world / video_to_4d_walkable_world already have notes+groups;
|
||||
# normalize formatting only.
|
||||
for name in ("video_to_4d_world.json", "video_to_4d_walkable_world.json"):
|
||||
save(name, load(name))
|
||||
print(f"[ok] reformat {name}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,543 @@
|
||||
"""Standalone CPU smoke test for the 4D-world node stack (no ComfyUI, no CUDA,
|
||||
no model downloads).
|
||||
|
||||
Run with:
|
||||
python notebooks/smoke_test_4d.py
|
||||
|
||||
Stubs `folder_paths` via sys.modules injection so the repo modules import
|
||||
outside the ComfyUI runtime, then functionally exercises the NEW code paths
|
||||
with small synthetic data:
|
||||
|
||||
1. interpolate_se3 (pointcloud_nodes, contract C1)
|
||||
2. render_gaussians (GS_nodes, contract C2) shapes + empty case
|
||||
3. render_gaussians fast anisotropic footprint
|
||||
4. GaussianSplats4D.at_time (GS4D_nodes, contract C3)
|
||||
5. BuildSplats4D kNN track binding
|
||||
6. SplitSplatsByMask
|
||||
7. MotionMaskFromDepth
|
||||
8. align_depth_scale (world_nodes, contract C4) + DepthEdgeFilter
|
||||
9. FuseSplats weighted voxel fusion
|
||||
10. SphereSplatSeed pano -> splat sphere -> render round-trip
|
||||
11. Fisheye image-circle exclusion (motion mask / tracks / splat split)
|
||||
"""
|
||||
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
import traceback
|
||||
import types
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Environment setup: repo on sys.path + folder_paths stub (before repo imports)
|
||||
# --------------------------------------------------------------------------- #
|
||||
REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
if REPO_ROOT not in sys.path:
|
||||
sys.path.insert(0, REPO_ROOT)
|
||||
|
||||
_TMP_DIR = tempfile.mkdtemp(prefix="smoke_test_4d_")
|
||||
|
||||
|
||||
def _stub_get_save_image_path(filename_prefix, output_dir, *args, **kwargs):
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
return output_dir, filename_prefix, 0, "", filename_prefix
|
||||
|
||||
|
||||
_fp_stub = types.ModuleType("folder_paths")
|
||||
_fp_stub.get_input_directory = lambda: _TMP_DIR
|
||||
_fp_stub.get_output_directory = lambda: _TMP_DIR
|
||||
_fp_stub.get_temp_directory = lambda: _TMP_DIR
|
||||
_fp_stub.get_save_image_path = _stub_get_save_image_path
|
||||
_fp_stub.get_annotated_filepath = lambda name: os.path.join(_TMP_DIR, name)
|
||||
_fp_stub.exists_annotated_filepath = lambda name: os.path.exists(os.path.join(_TMP_DIR, name))
|
||||
_fp_stub.get_filename_list = lambda folder: []
|
||||
_fp_stub.models_dir = _TMP_DIR
|
||||
sys.modules["folder_paths"] = _fp_stub
|
||||
|
||||
import numpy as np # noqa: E402
|
||||
import torch # noqa: E402
|
||||
|
||||
import GS_nodes # noqa: E402
|
||||
import GS4D_nodes # noqa: E402
|
||||
import pointcloud_nodes # noqa: E402
|
||||
import world_nodes # noqa: E402
|
||||
|
||||
GaussianSplats = GS_nodes.GaussianSplats
|
||||
|
||||
torch.manual_seed(0)
|
||||
np.random.seed(0)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Helpers
|
||||
# --------------------------------------------------------------------------- #
|
||||
def make_splats(
|
||||
xyz: torch.Tensor,
|
||||
sigma: float = 0.05,
|
||||
color: tuple = None,
|
||||
opacity_logit: float = 4.0,
|
||||
) -> GaussianSplats:
|
||||
"""Isotropic sh_order-0 splats at the given positions."""
|
||||
n = xyz.shape[0]
|
||||
if color is None:
|
||||
rgb = torch.rand(n, 3)
|
||||
else:
|
||||
rgb = torch.tensor(color, dtype=torch.float32).view(1, 3).expand(n, 3)
|
||||
C0 = 0.28209479177387814
|
||||
return GaussianSplats(
|
||||
xyz=xyz.float(),
|
||||
scale=torch.full((n, 3), math.log(sigma)),
|
||||
rotation=torch.tensor([1.0, 0.0, 0.0, 0.0]).view(1, 4).expand(n, 4).contiguous(),
|
||||
opacity=torch.full((n, 1), float(opacity_logit)),
|
||||
f_dc=((rgb - 0.5) / C0).contiguous(),
|
||||
f_rest=torch.zeros(n, 0),
|
||||
sh_order=0,
|
||||
)
|
||||
|
||||
|
||||
def rot_x(deg: float) -> torch.Tensor:
|
||||
a = math.radians(deg)
|
||||
return torch.tensor(
|
||||
[[1, 0, 0], [0, math.cos(a), -math.sin(a)], [0, math.sin(a), math.cos(a)]],
|
||||
dtype=torch.float32,
|
||||
)
|
||||
|
||||
|
||||
def rot_y(deg: float) -> torch.Tensor:
|
||||
a = math.radians(deg)
|
||||
return torch.tensor(
|
||||
[[math.cos(a), 0, math.sin(a)], [0, 1, 0], [-math.sin(a), 0, math.cos(a)]],
|
||||
dtype=torch.float32,
|
||||
)
|
||||
|
||||
|
||||
def make_pose(R: torch.Tensor, t) -> torch.Tensor:
|
||||
M = torch.eye(4)
|
||||
M[:3, :3] = R
|
||||
M[:3, 3] = torch.tensor(t, dtype=torch.float32)
|
||||
return M
|
||||
|
||||
|
||||
IDENTITY_4X4 = torch.eye(4)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Tests
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_01_interpolate_se3():
|
||||
poses = torch.stack(
|
||||
[
|
||||
make_pose(torch.eye(3), [0.0, 0.0, 0.0]),
|
||||
make_pose(rot_y(90.0), [1.0, 2.0, 3.0]),
|
||||
make_pose(rot_y(90.0) @ rot_x(45.0), [-1.0, 0.0, 2.0]),
|
||||
]
|
||||
)
|
||||
out = pointcloud_nodes.interpolate_se3(poses, 10)
|
||||
assert out.shape == (10, 4, 4), f"shape {tuple(out.shape)}"
|
||||
|
||||
eye = torch.eye(3)
|
||||
for i in range(10):
|
||||
R = out[i, :3, :3]
|
||||
ortho_err = (R @ R.T - eye).abs().max().item()
|
||||
det = torch.det(R).item()
|
||||
assert ortho_err < 1e-4, f"step {i}: R@R.T deviates from I by {ortho_err}"
|
||||
assert abs(det - 1.0) < 1e-4, f"step {i}: det(R)={det}"
|
||||
assert torch.allclose(out[i, 3], torch.tensor([0.0, 0.0, 0.0, 1.0]), atol=1e-6)
|
||||
|
||||
assert (out[0] - poses[0]).abs().max().item() < 1e-4, "start pose mismatch"
|
||||
assert (out[-1] - poses[-1]).abs().max().item() < 1e-4, "end pose mismatch"
|
||||
|
||||
# K == 1 repeats.
|
||||
rep = pointcloud_nodes.interpolate_se3(poses[:1], 5)
|
||||
assert rep.shape == (5, 4, 4)
|
||||
assert (rep - poses[0]).abs().max().item() < 1e-6
|
||||
|
||||
|
||||
def test_02_render_gaussians_shapes_and_empty():
|
||||
n, H, W = 200, 48, 64
|
||||
xyz = torch.stack(
|
||||
[
|
||||
torch.rand(n) * 2.0 - 1.0,
|
||||
torch.rand(n) * 2.0 - 1.0,
|
||||
torch.rand(n) * 3.0 + 2.0,
|
||||
],
|
||||
dim=-1,
|
||||
)
|
||||
splats = make_splats(xyz, sigma=0.05)
|
||||
|
||||
for projection, fov in (("PINHOLE", 90.0), ("EQUIRECTANGULAR", 360.0)):
|
||||
image, mask, disparity = GS_nodes.render_gaussians(
|
||||
splats, IDENTITY_4X4, projection, fov, W, H,
|
||||
render_mode="fast", device="cpu",
|
||||
)
|
||||
assert image.shape == (1, H, W, 3), f"{projection} image {tuple(image.shape)}"
|
||||
assert mask.shape == (H, W), f"{projection} mask {tuple(mask.shape)}"
|
||||
assert disparity.shape == (1, H, W, 1), f"{projection} disparity {tuple(disparity.shape)}"
|
||||
assert torch.isfinite(image).all() and torch.isfinite(disparity).all()
|
||||
assert float(mask.min()) >= 0.0 and float(mask.max()) <= 1.0 + 1e-6
|
||||
assert float(mask.sum()) > 0.0, f"{projection}: nothing rendered"
|
||||
|
||||
# Empty case: every splat strictly behind a pinhole camera (known past bug:
|
||||
# early return used to yield only 2 outputs).
|
||||
behind = make_splats(xyz * torch.tensor([1.0, 1.0, -1.0]), sigma=0.05)
|
||||
result = GS_nodes.render_gaussians(
|
||||
behind, IDENTITY_4X4, "PINHOLE", 90.0, W, H,
|
||||
render_mode="fast", device="cpu",
|
||||
)
|
||||
assert isinstance(result, tuple) and len(result) == 3, f"empty render returned {len(result)} outputs"
|
||||
image, mask, disparity = result
|
||||
assert image.shape == (1, H, W, 3)
|
||||
assert mask.shape == (H, W)
|
||||
assert disparity.shape == (1, H, W, 1)
|
||||
assert float(mask.sum()) == 0.0
|
||||
|
||||
|
||||
def test_03_fast_mode_anisotropy():
|
||||
H = W = 128
|
||||
ang = math.radians(45.0) / 2.0
|
||||
splats = GaussianSplats(
|
||||
xyz=torch.tensor([[0.0, 0.0, 3.0]]),
|
||||
scale=torch.log(torch.tensor([[0.5, 0.01, 0.01]])),
|
||||
rotation=torch.tensor([[math.cos(ang), 0.0, 0.0, math.sin(ang)]]), # 45 deg about +z
|
||||
opacity=torch.tensor([[6.0]]),
|
||||
f_dc=torch.zeros(1, 3),
|
||||
f_rest=torch.zeros(1, 0),
|
||||
sh_order=0,
|
||||
)
|
||||
image, mask, disparity = GS_nodes.render_gaussians(
|
||||
splats, IDENTITY_4X4, "PINHOLE", 60.0, W, H,
|
||||
render_mode="fast", max_radius=64, device="cpu",
|
||||
)
|
||||
assert float(mask.sum()) > 0.0, "elongated splat rendered nothing"
|
||||
|
||||
# Alpha-weighted pixel covariance of the footprint.
|
||||
ys, xs = torch.meshgrid(
|
||||
torch.arange(H, dtype=torch.float32), torch.arange(W, dtype=torch.float32),
|
||||
indexing="ij",
|
||||
)
|
||||
w = mask.flatten()
|
||||
wsum = w.sum()
|
||||
mx = (w * xs.flatten()).sum() / wsum
|
||||
my = (w * ys.flatten()).sum() / wsum
|
||||
dx = xs.flatten() - mx
|
||||
dy = ys.flatten() - my
|
||||
cxx = (w * dx * dx).sum() / wsum
|
||||
cyy = (w * dy * dy).sum() / wsum
|
||||
cxy = (w * dx * dy).sum() / wsum
|
||||
cov = torch.tensor([[cxx, cxy], [cxy, cyy]])
|
||||
evals, evecs = torch.linalg.eigh(cov)
|
||||
ratio = float(evals[1] / evals[0].clamp(min=1e-8))
|
||||
assert ratio > 2.0, f"footprint not elongated: eigenvalue ratio {ratio:.2f}"
|
||||
|
||||
# Principal axis should be near 45 degrees (rotation honored).
|
||||
major = evecs[:, 1]
|
||||
angle = math.degrees(math.atan2(float(major[1]), float(major[0]))) % 180.0
|
||||
assert abs(angle - 45.0) < 15.0, f"major axis at {angle:.1f} deg, expected ~45"
|
||||
|
||||
|
||||
def test_04_at_time():
|
||||
T = 5
|
||||
canonical = make_splats(torch.tensor([[0.0, 0.0, 2.0], [0.0, 1.0, 3.0]]))
|
||||
static = make_splats(torch.tensor([[5.0, 5.0, 5.0]]))
|
||||
start = torch.tensor([[0.0, 0.0, 2.0], [0.0, 1.0, 3.0]])
|
||||
end = torch.tensor([[1.0, 0.0, 2.0], [0.0, -1.0, 3.0]])
|
||||
ts = torch.linspace(0.0, 1.0, T)
|
||||
trajectories = torch.stack([start + (end - start) * t for t in ts]) # [5,2,3]
|
||||
|
||||
s4d = GS4D_nodes.GaussianSplats4D(
|
||||
static=static, canonical=canonical, trajectories=trajectories, times=ts,
|
||||
)
|
||||
|
||||
mid = s4d.at_time(0.5)
|
||||
assert len(mid) == 3, f"count {len(mid)} != dynamic+static (3)"
|
||||
# Concat order is [static, dynamic].
|
||||
assert torch.allclose(mid.xyz[0], static.xyz[0], atol=1e-6)
|
||||
expected_mid = 0.5 * (start + end)
|
||||
assert torch.allclose(mid.xyz[1:], expected_mid, atol=1e-5), (
|
||||
f"midpoint mismatch: {mid.xyz[1:]} vs {expected_mid}"
|
||||
)
|
||||
|
||||
lo = s4d.at_time(-1.0)
|
||||
hi = s4d.at_time(2.0)
|
||||
assert torch.allclose(lo.xyz[1:], start, atol=1e-5), "t<range should clamp to first step"
|
||||
assert torch.allclose(hi.xyz[1:], end, atol=1e-5), "t>range should clamp to last step"
|
||||
|
||||
|
||||
def test_05_build_splats4d():
|
||||
T = 5
|
||||
ts = torch.linspace(0.0, 1.0, T)
|
||||
# Two control tracks moving apart along x.
|
||||
track_a = torch.stack([torch.tensor([-1.0 - 2.0 * t, 0.0, 2.0]) for t in ts])
|
||||
track_b = torch.stack([torch.tensor([1.0 + 2.0 * t, 0.0, 2.0]) for t in ts])
|
||||
trajectories3d = torch.stack([track_a, track_b], dim=1) # [T,2,3]
|
||||
|
||||
canonical = make_splats(torch.tensor([[-1.05, 0.0, 2.0], [1.05, 0.0, 2.0]]))
|
||||
node = GS4D_nodes.BuildSplats4D()
|
||||
(s4d,) = node.build_splats4d(
|
||||
canonical=canonical,
|
||||
trajectories3d=trajectories3d,
|
||||
reference_index=0,
|
||||
knn=1,
|
||||
rbf_gamma=0.0,
|
||||
device="cpu",
|
||||
)
|
||||
traj = s4d.trajectories
|
||||
assert traj.shape == (T, 2, 3), f"trajectories shape {tuple(traj.shape)}"
|
||||
# Reference timestep: splats stay at their canonical positions.
|
||||
assert torch.allclose(traj[0], canonical.xyz, atol=1e-5)
|
||||
# Each splat follows its nearest track's displacement direction.
|
||||
disp0 = traj[-1, 0] - traj[0, 0]
|
||||
disp1 = traj[-1, 1] - traj[0, 1]
|
||||
assert disp0[0] < -1.0, f"splat 0 should move -x with track A, moved {disp0.tolist()}"
|
||||
assert disp1[0] > 1.0, f"splat 1 should move +x with track B, moved {disp1.tolist()}"
|
||||
assert torch.allclose(traj[-1, 0], torch.tensor([-3.05, 0.0, 2.0]), atol=1e-4)
|
||||
assert torch.allclose(traj[-1, 1], torch.tensor([3.05, 0.0, 2.0]), atol=1e-4)
|
||||
|
||||
|
||||
def test_06_split_splats_by_mask():
|
||||
H = W = 32
|
||||
mask = torch.zeros(H, W)
|
||||
mask[:, : W // 2] = 1.0 # left half white
|
||||
|
||||
# 10 splats projecting into the left half (x<0), 10 into the right half,
|
||||
# 5 behind the camera.
|
||||
jitter = torch.linspace(-0.1, 0.1, 10)
|
||||
left = torch.stack([torch.full((10,), -0.5) + jitter * 0.1, jitter, torch.full((10,), 2.0)], dim=-1)
|
||||
right = torch.stack([torch.full((10,), 0.5) + jitter * 0.1, jitter, torch.full((10,), 2.0)], dim=-1)
|
||||
behind = torch.stack([jitter[:5], jitter[:5], torch.full((5,), -2.0)], dim=-1)
|
||||
splats = make_splats(torch.cat([left, right, behind], dim=0))
|
||||
|
||||
node = GS4D_nodes.SplitSplatsByMask()
|
||||
inside, outside = node.split_splats(
|
||||
splats=splats,
|
||||
mask=mask,
|
||||
projection="PINHOLE",
|
||||
horizontal_fov=90.0,
|
||||
threshold=0.5,
|
||||
camera_matrix=None,
|
||||
device="cpu",
|
||||
)
|
||||
assert len(inside) == 10, f"inside count {len(inside)} != 10"
|
||||
assert len(outside) == 15, f"outside count {len(outside)} != 15 (10 right + 5 behind)"
|
||||
assert (inside.xyz[:, 0] < 0).all(), "inside splats should be the x<0 group"
|
||||
|
||||
|
||||
def test_07_motion_mask_from_depth():
|
||||
T, H, W = 6, 32, 32
|
||||
depth = torch.full((T, H, W), 5.0)
|
||||
r0, r1 = 8, 16
|
||||
for t in range(T):
|
||||
depth[t, r0:r1, r0:r1] = 3.0 + 0.4 * t # depth-changing square patch
|
||||
|
||||
poses = torch.eye(4).unsqueeze(0).expand(T, 4, 4).contiguous()
|
||||
node = GS4D_nodes.MotionMaskFromDepth()
|
||||
(mask,) = node.motion_mask(
|
||||
depth_seq=depth,
|
||||
trajectory=poses,
|
||||
input_projection="PINHOLE",
|
||||
input_horizontal_fov=90.0,
|
||||
threshold=0.10,
|
||||
frame_gap=2,
|
||||
dilate=0,
|
||||
device="cpu",
|
||||
)
|
||||
assert mask.shape == (T, H, W), f"mask shape {tuple(mask.shape)}"
|
||||
|
||||
patch = mask[:, r0:r1, r0:r1]
|
||||
background = mask.clone()
|
||||
background[:, r0:r1, r0:r1] = 0.0
|
||||
patch_mean = float(patch.mean())
|
||||
bg_sum = float(background.sum())
|
||||
assert patch_mean > 0.9, f"moving square under-detected: mean {patch_mean:.3f}"
|
||||
assert bg_sum == 0.0, f"static plane falsely flagged: {bg_sum} pixels"
|
||||
|
||||
|
||||
def test_08_align_depth_scale_and_depth_edge_filter():
|
||||
H = W = 32
|
||||
new_depth = torch.rand(H, W) * 9.0 + 1.0
|
||||
# ref disparity = 0.5 * new disparity + 0.1 (i.e. ref = 2*new before shift).
|
||||
true_scale, true_shift = 0.5, 0.1
|
||||
ref_depth = 1.0 / (true_scale / new_depth + true_shift)
|
||||
valid = torch.ones(H, W)
|
||||
|
||||
aligned, scale, shift = world_nodes.align_depth_scale(
|
||||
new_depth, ref_depth, valid, mode="scale_shift"
|
||||
)
|
||||
assert abs(scale - true_scale) / true_scale < 0.05, f"scale {scale} vs {true_scale}"
|
||||
assert abs(shift - true_shift) / true_shift < 0.05, f"shift {shift} vs {true_shift}"
|
||||
rel_err = float(((aligned - ref_depth).abs() / ref_depth).max())
|
||||
assert rel_err < 0.01, f"aligned depth off by {rel_err:.4f} (rel)"
|
||||
|
||||
# DepthEdgeFilter: a vertical step edge must be masked out, flat kept.
|
||||
depth = torch.full((H, W), 1.0)
|
||||
depth[:, W // 2 :] = 5.0
|
||||
node = pointcloud_nodes.DepthEdgeFilter()
|
||||
(valid_mask,) = node.filter_edges(depth, relative_threshold=0.05, dilate=1)
|
||||
assert valid_mask.shape == (H, W)
|
||||
edge_cols = valid_mask[:, W // 2 - 1 : W // 2 + 1]
|
||||
assert float(edge_cols.max()) == 0.0, "step-edge pixels not masked out"
|
||||
assert float(valid_mask[:, : W // 2 - 3].min()) == 1.0, "flat left region wrongly masked"
|
||||
assert float(valid_mask[:, W // 2 + 3 :].min()) == 1.0, "flat right region wrongly masked"
|
||||
|
||||
|
||||
def test_09_fuse_splats():
|
||||
n = 20
|
||||
voxel = 0.5
|
||||
base = torch.stack(
|
||||
[
|
||||
torch.arange(n, dtype=torch.float32) * voxel + 0.15,
|
||||
torch.full((n,), 0.15),
|
||||
torch.full((n,), 0.15),
|
||||
],
|
||||
dim=-1,
|
||||
)
|
||||
cloud_a = make_splats(base)
|
||||
cloud_b = make_splats(base + 0.2) # same voxels as A (0.15+0.2 < 0.5)
|
||||
|
||||
node = GS_nodes.FuseSplats()
|
||||
(fused,) = node.fuse_splats(cloud_a, cloud_b, voxel, "smart", 1.0, 1.0, device="cpu")
|
||||
assert len(fused) < len(cloud_a) + len(cloud_b), (
|
||||
f"voxel fuse did not reduce: {len(fused)} vs {len(cloud_a) + len(cloud_b)}"
|
||||
)
|
||||
assert len(fused) == n, f"expected one splat per voxel ({n}), got {len(fused)}"
|
||||
|
||||
# Strong weight_a pulls fused positions onto cloud A.
|
||||
(fused_w,) = node.fuse_splats(cloud_a, cloud_b, voxel, "average", 1000.0, 1.0, device="cpu")
|
||||
assert len(fused_w) == n
|
||||
d_a = torch.cdist(fused_w.xyz, cloud_a.xyz).min(dim=1).values
|
||||
d_b = torch.cdist(fused_w.xyz, cloud_b.xyz).min(dim=1).values
|
||||
assert float(d_a.max()) < 0.01, f"fused positions not near cloud A (max dist {float(d_a.max()):.4f})"
|
||||
assert (d_a < d_b).all(), "weight_a=1000 should pull fused splats toward cloud A"
|
||||
|
||||
|
||||
def test_10_sphere_splat_seed():
|
||||
H, W = 64, 128
|
||||
stride = 2
|
||||
color = (0.2, 0.6, 0.9)
|
||||
pano = torch.tensor(color).view(1, 1, 1, 3).expand(1, H, W, 3).contiguous()
|
||||
|
||||
node = world_nodes.SphereSplatSeed()
|
||||
(splats,) = node.seed_sphere(
|
||||
image=pano,
|
||||
horizontal_fov=360.0,
|
||||
radius=5.0,
|
||||
splat_scale_frac=1.5,
|
||||
stride=stride,
|
||||
device="cpu",
|
||||
)
|
||||
expected = (H // stride) * (W // stride)
|
||||
assert abs(len(splats) - expected) <= max(4, expected // 20), (
|
||||
f"splat count {len(splats)} far from expected ~{expected}"
|
||||
)
|
||||
|
||||
image, mask, disparity = GS_nodes.render_gaussians(
|
||||
splats, IDENTITY_4X4, "PINHOLE", 60.0, 64, 64,
|
||||
render_mode="fast", device="cpu",
|
||||
)
|
||||
assert float(mask.sum()) > 0.0, "pinhole render of the sphere seed is empty"
|
||||
solid = mask > 0.9
|
||||
assert bool(solid.any()), "no confidently covered pixels in the render"
|
||||
rendered = image[0][solid] # [K,3]
|
||||
target = torch.tensor(color)
|
||||
err = (rendered.mean(dim=0) - target).abs().max().item()
|
||||
assert err < 0.05, f"color round-trip failed: rendered mean {rendered.mean(dim=0).tolist()} vs {color}"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Runner
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_11_fisheye_circle_exclusion():
|
||||
"""Fisheye pixels/points outside the image circle (r > 1) must be ignored:
|
||||
corners never become 'dynamic', corner tracks never lift to 3D, and points
|
||||
at view angles beyond fov/2 are not matched against the mask."""
|
||||
T, H, W = 6, 32, 32
|
||||
fov = 180.0
|
||||
|
||||
# --- MotionMaskFromDepth: wildly flickering corners must stay static ----
|
||||
depth = torch.full((T, H, W), 5.0)
|
||||
depth[:, 14:18, 14:18] = torch.linspace(5.0, 2.0, T).view(T, 1, 1) # real motion
|
||||
u = torch.linspace(-1, 1, W).view(1, W).expand(H, W)
|
||||
v = torch.linspace(-1, 1, H).view(H, 1).expand(H, W)
|
||||
corners = (u * u + v * v) > 1.0 + 1e-6
|
||||
for t in range(T): # garbage depth flicker outside the circle
|
||||
depth[t][corners] = 1.0 if t % 2 == 0 else 10.0
|
||||
identity = torch.eye(4).unsqueeze(0).expand(T, 4, 4).contiguous()
|
||||
(dyn,) = GS4D_nodes.MotionMaskFromDepth().motion_mask(
|
||||
depth_seq=depth, trajectory=identity, input_projection="FISHEYE",
|
||||
input_horizontal_fov=fov, threshold=0.10, frame_gap=2, dilate=0,
|
||||
device="cpu",
|
||||
)
|
||||
assert dyn[:, corners].max().item() == 0.0, "fisheye corners were flagged dynamic"
|
||||
assert dyn[:, 14:18, 14:18].max().item() == 1.0, "real in-circle motion missed"
|
||||
|
||||
# --- TracksToTrajectories: corner track must be dropped -----------------
|
||||
tracks = torch.zeros(T, 2, 2)
|
||||
tracks[:, 0, 0] = W / 2.0 # center track
|
||||
tracks[:, 0, 1] = H / 2.0
|
||||
tracks[:, 1, 0] = 0.0 # corner track (u=v=-1, r=1.414)
|
||||
tracks[:, 1, 1] = 0.0
|
||||
visibility = torch.ones(T, 2)
|
||||
traj3d, track_ok = GS4D_nodes.TracksToTrajectories().tracks_to_trajectories(
|
||||
tracks=tracks, visibility=visibility, depth_seq=depth,
|
||||
input_projection="FISHEYE", input_horizontal_fov=fov,
|
||||
min_visible_frac=0.5, device="cpu",
|
||||
)
|
||||
assert bool(track_ok[0]), "in-circle track unexpectedly invalid"
|
||||
assert not bool(track_ok[1]), "out-of-circle corner track was lifted to 3D"
|
||||
|
||||
# --- SplitSplatsByMask: angle beyond fov/2 lands diagonally inside the
|
||||
# [-1,1] square (r=1.33, u=v~0.94) but must not be matched to the mask ---
|
||||
front = torch.tensor([[0.0, 0.0, 1.0]])
|
||||
theta = math.radians(120.0)
|
||||
behind_diag = torch.tensor([[
|
||||
math.sin(theta) * math.cos(math.radians(45.0)),
|
||||
math.sin(theta) * math.sin(math.radians(45.0)),
|
||||
math.cos(theta),
|
||||
]])
|
||||
splats = make_splats(torch.cat([front, behind_diag], dim=0))
|
||||
inside, outside = GS4D_nodes.SplitSplatsByMask().split_splats(
|
||||
splats=splats, mask=torch.ones(H, W), projection="FISHEYE",
|
||||
horizontal_fov=fov, threshold=0.5, device="cpu",
|
||||
)
|
||||
assert inside.xyz.shape[0] == 1, "expected only the in-fov splat inside"
|
||||
assert outside.xyz.shape[0] == 1, "beyond-fov splat must fall outside"
|
||||
|
||||
|
||||
TESTS = [
|
||||
test_01_interpolate_se3,
|
||||
test_02_render_gaussians_shapes_and_empty,
|
||||
test_03_fast_mode_anisotropy,
|
||||
test_04_at_time,
|
||||
test_05_build_splats4d,
|
||||
test_06_split_splats_by_mask,
|
||||
test_07_motion_mask_from_depth,
|
||||
test_08_align_depth_scale_and_depth_edge_filter,
|
||||
test_09_fuse_splats,
|
||||
test_10_sphere_splat_seed,
|
||||
test_11_fisheye_circle_exclusion,
|
||||
]
|
||||
|
||||
|
||||
def main() -> int:
|
||||
passed = 0
|
||||
failed = []
|
||||
for test in TESTS:
|
||||
name = test.__name__
|
||||
try:
|
||||
test()
|
||||
except Exception:
|
||||
failed.append(name)
|
||||
print(f"[FAIL] {name}")
|
||||
traceback.print_exc()
|
||||
else:
|
||||
passed += 1
|
||||
print(f"[ ok ] {name}")
|
||||
print(f"\n{passed}/{len(TESTS)} tests passed")
|
||||
if failed:
|
||||
print("Failed:", ", ".join(failed))
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,199 @@
|
||||
"""Offline tests for install.py logic and the flux-pack import resolution in
|
||||
flux_fisheye_filling_nodes.py. Stubs pip/git so nothing is actually installed.
|
||||
|
||||
Run: python notebooks/test_install_logic.py
|
||||
"""
|
||||
|
||||
import importlib.util
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
import types
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
PASS = []
|
||||
|
||||
|
||||
def ok(name):
|
||||
PASS.append(name)
|
||||
print(f"[ ok ] {name}")
|
||||
|
||||
|
||||
def load_install_module():
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"camera_install", os.path.join(REPO, "install.py")
|
||||
)
|
||||
mod = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(mod)
|
||||
return mod
|
||||
|
||||
|
||||
def stub_commands(mod):
|
||||
"""Replace subprocess-based helpers with recorders."""
|
||||
calls = []
|
||||
mod._run = lambda cmd, cwd=None: calls.append((tuple(cmd), cwd))
|
||||
mod._pip_install = lambda *args: calls.append((("pip",) + args, None))
|
||||
mod._git = lambda: "git"
|
||||
return calls
|
||||
|
||||
|
||||
def test_sharp_early_return():
|
||||
mod = load_install_module()
|
||||
calls = stub_commands(mod)
|
||||
# Build the "already materialized" state explicitly — the repo's own
|
||||
# submodule may not be checked out (e.g. CI clones without --recursive).
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
mod.NODE_DIR = os.path.join(tmp, "camera-comfyUI")
|
||||
os.makedirs(os.path.join(mod.NODE_DIR, "submodules", "ml-sharpt", "src", "sharp"))
|
||||
mod.ensure_sharp_checkout()
|
||||
assert calls == [], f"expected no commands, got {calls}"
|
||||
ok("sharp checkout present -> no git calls")
|
||||
|
||||
|
||||
def test_sharp_clone_fallback():
|
||||
mod = load_install_module()
|
||||
calls = stub_commands(mod)
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
mod.NODE_DIR = os.path.join(tmp, "custom_nodes", "camera-comfyUI")
|
||||
os.makedirs(mod.NODE_DIR)
|
||||
# No .git -> should go straight to direct clone
|
||||
mod.ensure_sharp_checkout()
|
||||
assert len(calls) == 1 and "clone" in calls[0][0], calls
|
||||
assert mod.SHARP_GIT_URL in calls[0][0], calls
|
||||
ok("sharp missing + no .git -> direct clone")
|
||||
|
||||
|
||||
def test_vggt_uses_https_git():
|
||||
mod = load_install_module()
|
||||
calls = stub_commands(mod)
|
||||
mod._importable = lambda name: False
|
||||
mod.ensure_vggt()
|
||||
assert calls == [(("pip", "vggt @ git+https://github.com/facebookresearch/vggt.git"), None)], calls
|
||||
ok("vggt -> pip install from git+https (no ssh)")
|
||||
|
||||
|
||||
def test_flux_pack_skip_outside_comfyui():
|
||||
mod = load_install_module()
|
||||
calls = stub_commands(mod)
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
mod.NODE_DIR = os.path.join(tmp, "somewhere", "camera-comfyUI")
|
||||
os.makedirs(mod.NODE_DIR)
|
||||
mod.ensure_flux_inpainting_pack()
|
||||
assert calls == [], calls
|
||||
ok("flux pack: skipped when parent is not custom_nodes")
|
||||
|
||||
|
||||
def test_flux_pack_detects_existing_and_clones_when_missing():
|
||||
mod = load_install_module()
|
||||
for existing in ("inpainting_flux", "ComfyUI-Flux-Inpainting", "ComfyUI-Flux-Inpainting-main"):
|
||||
calls = stub_commands(mod)
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
cn = os.path.join(tmp, "custom_nodes")
|
||||
mod.NODE_DIR = os.path.join(cn, "camera-comfyUI")
|
||||
os.makedirs(mod.NODE_DIR)
|
||||
os.makedirs(os.path.join(cn, existing))
|
||||
mod.ensure_flux_inpainting_pack()
|
||||
assert calls == [], f"{existing}: {calls}"
|
||||
ok("flux pack: all existing folder names detected, no clone")
|
||||
|
||||
calls = stub_commands(mod)
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
cn = os.path.join(tmp, "custom_nodes")
|
||||
mod.NODE_DIR = os.path.join(cn, "camera-comfyUI")
|
||||
os.makedirs(mod.NODE_DIR)
|
||||
mod.ensure_flux_inpainting_pack()
|
||||
assert len(calls) == 1 and "clone" in calls[0][0], calls
|
||||
assert calls[0][0][-1] == os.path.join(cn, "inpainting_flux"), calls
|
||||
ok("flux pack: missing -> cloned as custom_nodes/inpainting_flux")
|
||||
|
||||
|
||||
def test_example_inputs_copied():
|
||||
mod = load_install_module()
|
||||
stub_commands(mod)
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
cn = os.path.join(tmp, "ComfyUI", "custom_nodes")
|
||||
mod.NODE_DIR = os.path.join(cn, "camera-comfyUI")
|
||||
src = os.path.join(mod.NODE_DIR, "example_inputs")
|
||||
os.makedirs(src)
|
||||
with open(os.path.join(src, "example.jpg"), "w") as f:
|
||||
f.write("x")
|
||||
input_dir = os.path.join(tmp, "ComfyUI", "input")
|
||||
os.makedirs(input_dir)
|
||||
with open(os.path.join(input_dir, "existing.jpg"), "w") as f:
|
||||
f.write("keep me")
|
||||
mod.ensure_example_inputs()
|
||||
mod.ensure_example_inputs() # idempotent, no overwrite
|
||||
assert os.path.isfile(os.path.join(input_dir, "example.jpg"))
|
||||
assert open(os.path.join(input_dir, "existing.jpg")).read() == "keep me"
|
||||
ok("example inputs copied to ComfyUI input dir, no overwrite")
|
||||
|
||||
|
||||
def test_main_never_fails():
|
||||
mod = load_install_module()
|
||||
|
||||
def boom():
|
||||
raise RuntimeError("no network")
|
||||
|
||||
mod.STEPS = (("step-a", boom), ("step-b", boom))
|
||||
assert mod.main() == 0
|
||||
ok("main() returns 0 even when every step fails")
|
||||
|
||||
|
||||
def test_flux_import_resolution():
|
||||
"""Build a fake ComfyUI tree and check _import_flux_inpainting finds the
|
||||
pack under a dashed folder name and honors its relative imports."""
|
||||
if "PIL" not in sys.modules:
|
||||
try:
|
||||
import PIL # noqa: F401
|
||||
except ImportError:
|
||||
pil = types.ModuleType("PIL")
|
||||
pil.Image = types.SimpleNamespace()
|
||||
sys.modules["PIL"] = pil
|
||||
sys.modules["PIL.Image"] = types.ModuleType("PIL.Image")
|
||||
|
||||
tmp = tempfile.mkdtemp()
|
||||
try:
|
||||
cn = os.path.join(tmp, "ComfyUI", "custom_nodes")
|
||||
pack = os.path.join(cn, "campack")
|
||||
os.makedirs(pack)
|
||||
for fname in ("reprojection_nodes.py", "flux_fisheye_filling_nodes.py"):
|
||||
shutil.copy(os.path.join(REPO, fname), pack)
|
||||
open(os.path.join(pack, "__init__.py"), "w").close()
|
||||
|
||||
# Fake flux pack under a dashed (non-identifier) folder name with a
|
||||
# relative import, mirroring the real repo layout.
|
||||
flux = os.path.join(cn, "ComfyUI-Flux-Inpainting")
|
||||
os.makedirs(os.path.join(flux, "modules"))
|
||||
open(os.path.join(flux, "__init__.py"), "w").close()
|
||||
open(os.path.join(flux, "modules", "__init__.py"), "w").close()
|
||||
with open(os.path.join(flux, "modules", "load_util.py"), "w") as f:
|
||||
f.write("MARKER = 'loaded'\n")
|
||||
with open(os.path.join(flux, "nodes.py"), "w") as f:
|
||||
f.write(
|
||||
"from .modules.load_util import MARKER\n"
|
||||
"class FluxNF4Inpainting:\n"
|
||||
" marker = MARKER\n"
|
||||
)
|
||||
|
||||
sys.path.insert(0, os.path.dirname(pack))
|
||||
mod = importlib.import_module("campack.flux_fisheye_filling_nodes")
|
||||
assert mod._flux_import_error is None, mod._flux_import_error
|
||||
assert mod.FluxInpainting is not None
|
||||
assert mod.FluxInpainting.marker == "loaded"
|
||||
ok("flux import: dashed folder name resolved incl. relative imports")
|
||||
finally:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
test_sharp_early_return()
|
||||
test_sharp_clone_fallback()
|
||||
test_vggt_uses_https_git()
|
||||
test_flux_pack_skip_outside_comfyui()
|
||||
test_flux_pack_detects_existing_and_clones_when_missing()
|
||||
test_example_inputs_copied()
|
||||
test_main_never_fails()
|
||||
test_flux_import_resolution()
|
||||
print(f"\n{len(PASS)} install-logic checks passed")
|
||||
@@ -0,0 +1,133 @@
|
||||
"""Validate workflows/*.json against the current node definitions.
|
||||
|
||||
For every camera-comfyUI node used in a workflow, compares the stored
|
||||
widgets_values against the widget list derived from the node's INPUT_TYPES
|
||||
(required + optional, in order, counting only widget-type inputs). A length
|
||||
mismatch means the workflow predates a node-schema change and will load with
|
||||
silently shifted/défault values.
|
||||
|
||||
Run: python notebooks/validate_workflows.py
|
||||
Exit code 1 if any mismatch is found (missing node types are also reported).
|
||||
"""
|
||||
|
||||
import glob
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
import types
|
||||
|
||||
REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
if REPO_ROOT not in sys.path:
|
||||
sys.path.insert(0, REPO_ROOT)
|
||||
|
||||
_TMP_DIR = tempfile.mkdtemp(prefix="wf_validate_")
|
||||
|
||||
|
||||
def _stub_get_save_image_path(filename_prefix, output_dir, *args, **kwargs):
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
return output_dir, filename_prefix, 0, "", filename_prefix
|
||||
|
||||
|
||||
_fp = types.ModuleType("folder_paths")
|
||||
_fp.get_input_directory = lambda: _TMP_DIR
|
||||
_fp.get_output_directory = lambda: _TMP_DIR
|
||||
_fp.get_temp_directory = lambda: _TMP_DIR
|
||||
_fp.get_save_image_path = _stub_get_save_image_path
|
||||
_fp.get_annotated_filepath = lambda name: os.path.join(_TMP_DIR, name)
|
||||
_fp.exists_annotated_filepath = lambda name: os.path.exists(os.path.join(_TMP_DIR, name))
|
||||
_fp.get_filename_list = lambda folder: []
|
||||
_fp.models_dir = _TMP_DIR
|
||||
sys.modules["folder_paths"] = _fp
|
||||
|
||||
# Light stubs for heavy deps that some modules import at module level but that
|
||||
# INPUT_TYPES itself does not need.
|
||||
for name in ("transformers", "diffusers"):
|
||||
if name not in sys.modules:
|
||||
try:
|
||||
__import__(name)
|
||||
except ImportError:
|
||||
stub = types.ModuleType(name)
|
||||
stub.pipeline = lambda *a, **k: None
|
||||
sys.modules[name] = stub
|
||||
|
||||
# Import the repo as a package so modules with relative imports load too.
|
||||
import importlib.util # noqa: E402
|
||||
|
||||
_spec = importlib.util.spec_from_file_location(
|
||||
"camcomfy",
|
||||
os.path.join(REPO_ROOT, "__init__.py"),
|
||||
submodule_search_locations=[REPO_ROOT],
|
||||
)
|
||||
_pkg = importlib.util.module_from_spec(_spec)
|
||||
sys.modules["camcomfy"] = _pkg
|
||||
_spec.loader.exec_module(_pkg)
|
||||
NODE_CLASS_MAPPINGS = dict(_pkg.NODE_CLASS_MAPPINGS)
|
||||
|
||||
WIDGET_TYPES = {"INT", "FLOAT", "STRING", "BOOLEAN"}
|
||||
|
||||
|
||||
def widget_specs(cls):
|
||||
"""Ordered (name, combo_options|None) for a node's widget inputs."""
|
||||
it = cls.INPUT_TYPES()
|
||||
specs = []
|
||||
for section in ("required", "optional"):
|
||||
for name, spec in it.get(section, {}).items():
|
||||
t = spec[0] if isinstance(spec, (tuple, list)) and spec else spec
|
||||
if isinstance(t, (list, tuple)): # combo box (either sequence type)
|
||||
specs.append((name, list(t)))
|
||||
elif isinstance(t, str) and t in WIDGET_TYPES:
|
||||
specs.append((name, None))
|
||||
# ComfyUI appends a control_after_generate widget after seeds
|
||||
if t == "INT" and name in ("seed", "noise_seed"):
|
||||
specs.append((f"{name}:control_after_generate", None))
|
||||
return specs
|
||||
|
||||
|
||||
def main() -> int:
|
||||
problems = 0
|
||||
for path in sorted(glob.glob(os.path.join(REPO_ROOT, "workflows", "**", "*.json"),
|
||||
recursive=True)):
|
||||
data = json.load(open(path, encoding="utf-8"))
|
||||
header_shown = False
|
||||
|
||||
def report(msg):
|
||||
nonlocal header_shown, problems
|
||||
if not header_shown:
|
||||
print(f"\n=== {os.path.basename(path)}")
|
||||
header_shown = True
|
||||
print(f" {msg}")
|
||||
problems += 1
|
||||
|
||||
for n in data.get("nodes", []):
|
||||
t = n["type"]
|
||||
if t not in NODE_CLASS_MAPPINGS:
|
||||
continue # builtin or third-party node
|
||||
specs = widget_specs(NODE_CLASS_MAPPINGS[t])
|
||||
got = n.get("widgets_values") or []
|
||||
if isinstance(got, dict):
|
||||
continue # API-style dict widgets (third-party save format)
|
||||
if len(got) != len(specs):
|
||||
report(
|
||||
f"N{n['id']} {t}: {len(got)} widget values, node now has "
|
||||
f"{len(specs)} widgets {[s[0] for s in specs]}; "
|
||||
f"stored={json.dumps(got)[:100]}"
|
||||
)
|
||||
continue
|
||||
for (name, options), value in zip(specs, got):
|
||||
# empty option lists are dynamic file dropdowns (input dir
|
||||
# listing) — not verifiable outside a real ComfyUI install
|
||||
if options and value not in options:
|
||||
report(
|
||||
f"N{n['id']} {t}: widget '{name}' has stale value "
|
||||
f"{value!r}, valid options are {options}"
|
||||
)
|
||||
if problems:
|
||||
print(f"\n{problems} mismatches found")
|
||||
return 1
|
||||
print("all workflow widget schemas match current node definitions")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -7,7 +7,10 @@ import os
|
||||
import folder_paths
|
||||
import logging
|
||||
import hashlib
|
||||
from kornia.filters import median_blur
|
||||
try:
|
||||
from kornia.filters import median_blur
|
||||
except ImportError: # kornia is optional; median_blur is not used in this module
|
||||
median_blur = None
|
||||
|
||||
from tqdm import tqdm
|
||||
# Try importing open3d and its visualization modules; log a warning if not found
|
||||
@@ -113,7 +116,7 @@ def XYZ_to_equirect(X: torch.Tensor, Y: torch.Tensor, Z: torch.Tensor, fov: floa
|
||||
Convert XYZ coordinates to normalized UV and depth using equirectangular projection.
|
||||
"""
|
||||
# full 360°×180°
|
||||
fov_rad = math.radians(fov)
|
||||
fov_rad = math.radians(fov) / 2
|
||||
depth = torch.sqrt(X**2 + Y**2 + Z**2)
|
||||
lon = torch.atan2(X, Z) # –π → +π
|
||||
lat = torch.asin(Y / depth) # –π/2 → +π/2
|
||||
@@ -136,6 +139,116 @@ def project_first_hit(volume_sparse: torch.Tensor) -> Tuple[torch.Tensor, torch.
|
||||
|
||||
return rgba.permute(2, 0, 1), first_hit.any(dim=2)
|
||||
|
||||
# ==== SE(3) trajectory interpolation ==== #
|
||||
def _rotmat_to_quat_wxyz(R: torch.Tensor) -> torch.Tensor:
|
||||
"""
|
||||
Convert a batch of rotation matrices [K,3,3] to unit quaternions [K,4] (wxyz).
|
||||
Uses Shepperd's method for numerical robustness. K is expected to be small
|
||||
(trajectory waypoints), so a Python loop is acceptable.
|
||||
"""
|
||||
quats = []
|
||||
for i in range(R.shape[0]):
|
||||
m = R[i]
|
||||
trace = m[0, 0] + m[1, 1] + m[2, 2]
|
||||
if trace > 0.0:
|
||||
s = torch.sqrt(trace + 1.0) * 2.0
|
||||
w = 0.25 * s
|
||||
x = (m[2, 1] - m[1, 2]) / s
|
||||
y = (m[0, 2] - m[2, 0]) / s
|
||||
z = (m[1, 0] - m[0, 1]) / s
|
||||
elif m[0, 0] > m[1, 1] and m[0, 0] > m[2, 2]:
|
||||
s = torch.sqrt(1.0 + m[0, 0] - m[1, 1] - m[2, 2]) * 2.0
|
||||
w = (m[2, 1] - m[1, 2]) / s
|
||||
x = 0.25 * s
|
||||
y = (m[0, 1] + m[1, 0]) / s
|
||||
z = (m[0, 2] + m[2, 0]) / s
|
||||
elif m[1, 1] > m[2, 2]:
|
||||
s = torch.sqrt(1.0 + m[1, 1] - m[0, 0] - m[2, 2]) * 2.0
|
||||
w = (m[0, 2] - m[2, 0]) / s
|
||||
x = (m[0, 1] + m[1, 0]) / s
|
||||
y = 0.25 * s
|
||||
z = (m[1, 2] + m[2, 1]) / s
|
||||
else:
|
||||
s = torch.sqrt(1.0 + m[2, 2] - m[0, 0] - m[1, 1]) * 2.0
|
||||
w = (m[1, 0] - m[0, 1]) / s
|
||||
x = (m[0, 2] + m[2, 0]) / s
|
||||
y = (m[1, 2] + m[2, 1]) / s
|
||||
z = 0.25 * s
|
||||
quats.append(torch.stack([w, x, y, z]))
|
||||
q = torch.stack(quats, dim=0)
|
||||
return q / q.norm(dim=-1, keepdim=True).clamp(min=1e-12)
|
||||
|
||||
|
||||
def _quat_wxyz_to_rotmat(q: torch.Tensor) -> torch.Tensor:
|
||||
"""Convert unit quaternions [N,4] (wxyz) to rotation matrices [N,3,3]."""
|
||||
q = q / q.norm(dim=-1, keepdim=True).clamp(min=1e-12)
|
||||
w, x, y, z = q.unbind(-1)
|
||||
R = torch.stack([
|
||||
1 - 2 * (y * y + z * z), 2 * (x * y - w * z), 2 * (x * z + w * y),
|
||||
2 * (x * y + w * z), 1 - 2 * (x * x + z * z), 2 * (y * z - w * x),
|
||||
2 * (x * z - w * y), 2 * (y * z + w * x), 1 - 2 * (x * x + y * y),
|
||||
], dim=-1).reshape(*q.shape[:-1], 3, 3)
|
||||
return R
|
||||
|
||||
|
||||
def _quat_slerp(q0: torch.Tensor, q1: torch.Tensor, alpha: torch.Tensor) -> torch.Tensor:
|
||||
"""
|
||||
Spherical linear interpolation between quaternion batches q0, q1 [N,4] (wxyz)
|
||||
with per-element interpolation factors alpha [N]. Falls back to normalized
|
||||
lerp when the quaternions are nearly parallel.
|
||||
"""
|
||||
dot = (q0 * q1).sum(dim=-1, keepdim=True)
|
||||
q1 = torch.where(dot < 0.0, -q1, q1) # shortest arc
|
||||
dot = dot.abs().clamp(max=1.0)
|
||||
a = alpha.reshape(-1, 1).to(q0.dtype)
|
||||
theta = torch.acos(dot)
|
||||
sin_theta = torch.sin(theta)
|
||||
near_parallel = sin_theta < 1e-6
|
||||
denom = sin_theta.clamp(min=1e-12)
|
||||
w0 = torch.where(near_parallel, 1.0 - a, torch.sin((1.0 - a) * theta) / denom)
|
||||
w1 = torch.where(near_parallel, a, torch.sin(a * theta) / denom)
|
||||
q = w0 * q0 + w1 * q1
|
||||
return q / q.norm(dim=-1, keepdim=True).clamp(min=1e-12)
|
||||
|
||||
|
||||
def interpolate_se3(trajectory: torch.Tensor, num_steps: int) -> torch.Tensor:
|
||||
"""trajectory [K,4,4] -> [num_steps,4,4]. Piecewise: quaternion SLERP on R, lerp on t.
|
||||
K==1 -> repeat. Must return valid rotation matrices (orthonormal)."""
|
||||
if isinstance(trajectory, np.ndarray):
|
||||
trajectory = torch.from_numpy(trajectory)
|
||||
trajectory = trajectory.float()
|
||||
if trajectory.dim() == 2:
|
||||
trajectory = trajectory.unsqueeze(0)
|
||||
if trajectory.dim() != 3 or trajectory.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"interpolate_se3 expects trajectory of shape [K,4,4], got {tuple(trajectory.shape)}")
|
||||
if num_steps < 1:
|
||||
raise ValueError(f"interpolate_se3 requires num_steps >= 1, got {num_steps}")
|
||||
K = trajectory.shape[0]
|
||||
if K == 1:
|
||||
return trajectory.expand(num_steps, 4, 4).clone()
|
||||
|
||||
R = trajectory[:, :3, :3]
|
||||
t = trajectory[:, :3, 3]
|
||||
q = _rotmat_to_quat_wxyz(R)
|
||||
# Enforce hemisphere continuity along the waypoint sequence so piecewise
|
||||
# SLERP always takes the shortest arc between consecutive poses.
|
||||
for k in range(1, K):
|
||||
if (q[k] * q[k - 1]).sum() < 0.0:
|
||||
q[k] = -q[k]
|
||||
|
||||
idxs = torch.linspace(0, K - 1, num_steps, device=trajectory.device)
|
||||
lower = idxs.floor().long().clamp(max=K - 2)
|
||||
upper = lower + 1
|
||||
alpha = (idxs - lower.float())
|
||||
|
||||
q_interp = _quat_slerp(q[lower], q[upper], alpha)
|
||||
t_interp = t[lower] * (1.0 - alpha).unsqueeze(-1) + t[upper] * alpha.unsqueeze(-1)
|
||||
|
||||
out = torch.eye(4, dtype=trajectory.dtype, device=trajectory.device).repeat(num_steps, 1, 1)
|
||||
out[:, :3, :3] = _quat_wxyz_to_rotmat(q_interp)
|
||||
out[:, :3, 3] = t_interp
|
||||
return out
|
||||
|
||||
# ==== Node Definitions ==== #
|
||||
class DepthToPointCloud:
|
||||
"""
|
||||
@@ -167,7 +280,7 @@ class DepthToPointCloud:
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("pointcloud",)
|
||||
FUNCTION = "depth_to_pointcloud"
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/PointCloud"
|
||||
|
||||
def depth_to_pointcloud(
|
||||
self,
|
||||
@@ -273,7 +386,7 @@ class TransformPointCloud:
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("transformed pointcloud",)
|
||||
FUNCTION = "transform_pointcloud"
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/PointCloud"
|
||||
|
||||
def transform_pointcloud(
|
||||
self,
|
||||
@@ -321,124 +434,169 @@ class ProjectPointCloud:
|
||||
RETURN_TYPES = ("IMAGE", "MASK", "TENSOR")
|
||||
RETURN_NAMES = ("image", "mask", "depth")
|
||||
FUNCTION = "project_pointcloud"
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/PointCloud"
|
||||
|
||||
def project_pointcloud(
|
||||
self,
|
||||
pointcloud: torch.Tensor,
|
||||
pointcloud: torch.Tensor,
|
||||
output_projection: str,
|
||||
output_horizontal_fov: float,
|
||||
output_width: int,
|
||||
output_width: int,
|
||||
output_height: int,
|
||||
point_size: int = 1,
|
||||
point_size: int = 1,
|
||||
return_inverse_depth: bool = False,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
"""
|
||||
Projects an (N×6) XYZRGB point cloud into an image,
|
||||
fills occlusion holes robustly, and returns:
|
||||
• img: [1,H,W,3] RGB image
|
||||
• mask: [H,W] foreground mask
|
||||
• depth: [1,H,W,1] depth (or inverse depth)
|
||||
"""
|
||||
device = pointcloud.device
|
||||
coords = pointcloud[:, :3]
|
||||
colors = pointcloud[:, 3:].float()
|
||||
xyz, rgb_raw = pointcloud[:, :3], pointcloud[:, 3:6].float()
|
||||
|
||||
# 1) Filter points in front of the camera
|
||||
mask_front = coords[:, 2] > 0
|
||||
coords = coords[mask_front]
|
||||
colors = colors[mask_front]
|
||||
# 1) Keep only points in front of camera
|
||||
in_front = xyz[:, 2] > 0
|
||||
xyz, rgb_raw = xyz[in_front], rgb_raw[in_front]
|
||||
|
||||
# 2) Project to normalized UV + depth
|
||||
X, Y, Z = coords.unbind(1)
|
||||
X, Y, Z = xyz.unbind(1)
|
||||
if output_projection == "PINHOLE":
|
||||
u, v, depth = XYZ_to_pinhole(X, Y, Z, output_horizontal_fov)
|
||||
u, v, d = XYZ_to_pinhole(X, Y, Z, output_horizontal_fov)
|
||||
elif output_projection == "FISHEYE":
|
||||
u, v, depth = XYZ_to_fisheye(X, Y, Z, output_horizontal_fov)
|
||||
u, v, d = XYZ_to_fisheye(X, Y, Z, output_horizontal_fov)
|
||||
else:
|
||||
u, v, depth = XYZ_to_equirect(X, Y, Z, output_horizontal_fov)
|
||||
u, v, d = XYZ_to_equirect(X, Y, Z, output_horizontal_fov)
|
||||
|
||||
# 3) Rasterize to pixel indices
|
||||
px = (u * (output_width - 1) / 2) + (output_width - 1) / 2
|
||||
py = (v * (output_height - 1) / 2) + (output_height - 1) / 2
|
||||
ix = px.round().clamp(0, output_width - 1).long()
|
||||
iy = py.round().clamp(0, output_height - 1).long()
|
||||
pix = iy * output_width + ix
|
||||
M = output_width * output_height
|
||||
W, H = output_width, output_height
|
||||
ix = ((u * 0.5 + 0.5) * (W - 1)).round().clamp(0, W - 1).long()
|
||||
iy = ((v * 0.5 + 0.5) * (H - 1)).round().clamp(0, H - 1).long()
|
||||
pix = iy * W + ix
|
||||
valid = (pix >= 0) & (pix < W * H)
|
||||
pix, d, rgb_raw = pix[valid], d[valid], rgb_raw[valid]
|
||||
|
||||
# —— NEW: drop any invalid / NaN→int_min projections ——
|
||||
valid = (pix >= 0) & (pix < M)
|
||||
depth = depth[valid]
|
||||
colors = colors[valid]
|
||||
pix = pix[valid]
|
||||
order = torch.arange(depth.size(0), device=device)
|
||||
# rebuild your "order" to match
|
||||
M = W * H
|
||||
# 3a) Front‐layer (nearest) depth
|
||||
z1 = torch.full((M,), float('inf'), device=device)
|
||||
z1.scatter_reduce_(0, pix, d, reduce='amin', include_self=True)
|
||||
|
||||
# 4) Allocate or reuse buffers
|
||||
if not hasattr(self, '_z_front') or self._z_front.numel() != M:
|
||||
self._z_front = torch.empty((M,), device=device)
|
||||
self._z_back = torch.empty((M,), device=device)
|
||||
self._idx = torch.full((M,), -1, dtype=torch.long, device=device)
|
||||
self._flat = torch.zeros((M, 4), device=device)
|
||||
z_front = self._z_front
|
||||
z_back = self._z_back
|
||||
idxbuf = self._idx
|
||||
flat = self._flat
|
||||
# 3b) Second‐layer (background) depth
|
||||
farther = d > z1[pix]
|
||||
pix2, d2 = pix[farther], d[farther]
|
||||
z2 = torch.full((M,), float('inf'), device=device)
|
||||
z2.scatter_reduce_(0, pix2, d2, reduce='amin', include_self=True)
|
||||
|
||||
# 5) Front z-buffer pass (nearest)
|
||||
z_front.fill_(float('inf'))
|
||||
z_front.scatter_reduce_(0, pix, depth, reduce='amin', include_self=True)
|
||||
sel_front = depth == z_front[pix]
|
||||
order = torch.arange(depth.size(0), device=device)
|
||||
order_m = torch.where(sel_front, order, depth.size(0))
|
||||
idxbuf.fill_(depth.size(0))
|
||||
idxbuf.scatter_reduce_(0, pix, order_m, reduce='amin', include_self=True)
|
||||
win_front = order == idxbuf[pix]
|
||||
# 3c) Foreground colour (from z1)
|
||||
keep = d == z1[pix]
|
||||
rgb = torch.zeros((M, 3), device=device)
|
||||
rgb[pix[keep]] = rgb_raw[keep].clamp(0, 255)
|
||||
|
||||
flat.fill_(0)
|
||||
flat[pix[win_front]] = colors[win_front]
|
||||
img4 = flat.view(output_height, output_width, 4)
|
||||
rgb = img4[..., :3].clamp(0, 255)
|
||||
alpha = (img4[..., 3] > 0).float()
|
||||
rgb *= alpha.unsqueeze(-1)
|
||||
depth_img = z_front.view(output_height, output_width)
|
||||
rgb_HR = rgb
|
||||
# reshape to image
|
||||
rgb = rgb.view(H, W, 3)
|
||||
z1 = z1.view(H, W)
|
||||
z2 = z2.view(H, W)
|
||||
fg_mask = z1 < float('inf') # has front hit
|
||||
occl = (z2 < float('inf')) # has any back hit
|
||||
rear_only = occl & ~fg_mask # true holes
|
||||
|
||||
# 6) Back z-buffer pass (farthest) for hole-filling
|
||||
# ── Iterative ring‐based in‐painting of rear‐only pixels ─────────────────────
|
||||
ker3 = torch.ones((1,1,3,3), device=device)
|
||||
ker3c = ker3.repeat(3,1,1,1)
|
||||
for _ in range(max(W, H)):
|
||||
# find rear_only pixels adjacent to current FG
|
||||
neigh = (
|
||||
F.max_pool2d(fg_mask.float()[None,None], 3, 1, 1).bool()[0,0]
|
||||
& ~fg_mask
|
||||
)
|
||||
to_fill = rear_only & neigh
|
||||
if not to_fill.any():
|
||||
break
|
||||
|
||||
# average depth + colour from current FG frontier
|
||||
d_t = z1.masked_fill(~fg_mask, 0)[None,None]
|
||||
c_t = rgb.permute(2,0,1)[None] # [1,3,H,W]
|
||||
m_t = fg_mask.float()[None,None]
|
||||
|
||||
sum_d = F.conv2d(d_t * m_t, ker3, padding=1)
|
||||
cnt_d = F.conv2d(m_t, ker3, padding=1).clamp(min=1)
|
||||
sum_c = F.conv2d(c_t * m_t, ker3c, padding=1, groups=3)
|
||||
cnt_c = cnt_d.repeat(1,3,1,1)
|
||||
|
||||
avg_d = (sum_d / cnt_d).squeeze()
|
||||
avg_c = (sum_c / cnt_c).squeeze().permute(1,2,0)
|
||||
|
||||
z1[to_fill] = avg_d[to_fill]
|
||||
rgb[to_fill] = avg_c[to_fill]
|
||||
fg_mask[to_fill] = True
|
||||
rear_only[to_fill] = False
|
||||
|
||||
# ── Depth‐aware generic hole closure ─────────────────────────────────────────
|
||||
# close_rad: radius of hole to close; depth_eps: depth jump tolerance
|
||||
close_rad = max(1, point_size // 2)
|
||||
depth_eps = 0.015
|
||||
pad = close_rad
|
||||
k = 2 * close_rad + 1
|
||||
ker = torch.ones((1,1,k,k), device=device)
|
||||
kerc = ker.repeat(3,1,1,1)
|
||||
front_t = fg_mask.float()[None,None]
|
||||
|
||||
# binary closing: dilate then erode
|
||||
D = F.max_pool2d(front_t, k, 1, pad)
|
||||
E = 1 - F.max_pool2d(1 - D, k, 1, pad)
|
||||
small_hole = E[0,0].bool() & ~fg_mask
|
||||
if small_hole.any():
|
||||
# compute local mean depth of FG
|
||||
z_t = z1.masked_fill(~fg_mask, 0)[None,None]
|
||||
cnt = F.conv2d(front_t, ker, padding=pad).clamp(min=1)
|
||||
z_avg = (F.conv2d(z_t, ker, padding=pad) / cnt)[0,0]
|
||||
|
||||
# depth‐range test
|
||||
z_near = F.max_pool2d(z1[None,None], 3,1,1)[0,0]
|
||||
z_far = -F.max_pool2d(-z1[None,None],3,1,1)[0,0]
|
||||
flat = (z_far - z_near) / z_avg.clamp(min=1e-6) < depth_eps
|
||||
|
||||
final = small_hole & flat
|
||||
if final.any():
|
||||
sum_d = F.conv2d(z_t, ker, padding=pad)
|
||||
sum_c = F.conv2d(rgb.permute(2,0,1)[None] * front_t, kerc,
|
||||
padding=pad, groups=3)
|
||||
avg_d = (sum_d / cnt)[0,0]
|
||||
avg_c = (sum_c / cnt.repeat(1,3,1,1))[0].permute(1,2,0)
|
||||
|
||||
z1[final] = avg_d[final]
|
||||
rgb[final] = avg_c[final]
|
||||
fg_mask[final] = True
|
||||
|
||||
# ── Optional morphological blur for larger point_size ───────────────────────
|
||||
if point_size > 1:
|
||||
z_back.fill_(-float('inf'))
|
||||
z_back.scatter_reduce_(0, pix, depth, reduce='amax', include_self=True)
|
||||
sel_back = depth == z_back[pix]
|
||||
order_m = torch.where(sel_back, order, -1)
|
||||
idxbuf.fill_(-1)
|
||||
idxbuf.scatter_reduce_(0, pix, order_m, reduce='amax', include_self=True)
|
||||
win_back = idxbuf[pix] >= 0
|
||||
r = point_size // 2
|
||||
k = 2 * r + 1
|
||||
pad = r
|
||||
ker = torch.ones((1,1,k,k), device=device)
|
||||
kerc = ker.repeat(3,1,1,1)
|
||||
d_t = z1[None,None]
|
||||
c_t = rgb.permute(2,0,1)[None]
|
||||
m_t = fg_mask.float()[None,None]
|
||||
|
||||
flat.fill_(0)
|
||||
flat[pix[win_back]] = colors[win_back]
|
||||
back4 = flat.view(output_height, output_width, 4)
|
||||
rgb_back = back4[..., :3].clamp(0,255)
|
||||
alpha_back = (back4[..., 3] > 0).float()
|
||||
z1 = (F.conv2d(d_t * m_t, ker, padding=pad) /
|
||||
F.conv2d(m_t, ker, padding=pad).clamp(min=1)).squeeze()
|
||||
rgb = (F.conv2d(c_t * m_t, kerc, padding=pad, groups=3) /
|
||||
F.conv2d(m_t, ker, padding=pad).repeat(1,3,1,1).clamp(min=1)
|
||||
).squeeze().permute(1,2,0)
|
||||
|
||||
# fill holes where front missed
|
||||
hole = (alpha == 0) & (alpha_back > 0)
|
||||
rgb[hole] = rgb_back[hole]
|
||||
alpha[hole] = 1.0
|
||||
depth_img[hole] = z_back.view(output_height, output_width)[hole]
|
||||
|
||||
# 7) Median-filter _only_ in hole regions
|
||||
if hole.any():
|
||||
# prepare for kornia median_blur: [B,C,H,W]
|
||||
rgb_t = rgb.permute(2,0,1).unsqueeze(0) # [1,3,H,W]
|
||||
# apply median filter
|
||||
rgb_med = median_blur(rgb_t, (point_size, point_size))
|
||||
# back to HWC
|
||||
rgb_med = rgb_med.squeeze(0).permute(1,2,0)
|
||||
# merge only at hole locations
|
||||
rgb[hole] = rgb_med[hole]
|
||||
# alpha already set to 1.0 for holes
|
||||
|
||||
# 8) Pack and return with original script shapes
|
||||
img = rgb.unsqueeze(0) # [1,H,W,3]
|
||||
mask_out = alpha # [H,W]
|
||||
depth4 = depth_img.unsqueeze(0).unsqueeze(-1) # [1,H,W,1]
|
||||
# 10) Pack outputs
|
||||
img = rgb.unsqueeze(0) # [1,H,W,3]
|
||||
mask = fg_mask.float() # [H,W]
|
||||
depth = z1.unsqueeze(0).unsqueeze(-1) # [1,H,W,1]
|
||||
if return_inverse_depth:
|
||||
depth4 = 1.0 / depth4.clamp(min=1e-6)
|
||||
depth4 = depth4 * mask_out.unsqueeze(0).unsqueeze(-1)
|
||||
return img, mask_out, depth4
|
||||
depth = 1.0 / depth.clamp(min=1e-6)
|
||||
depth *= mask.unsqueeze(0).unsqueeze(-1)
|
||||
return img, mask, depth
|
||||
|
||||
|
||||
|
||||
|
||||
class PointCloudUnion:
|
||||
"""
|
||||
@@ -456,7 +614,7 @@ class PointCloudUnion:
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES =("merged pointcloud",)
|
||||
FUNCTION = "union_pointclouds"
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/PointCloud"
|
||||
|
||||
def union_pointclouds(
|
||||
self,
|
||||
@@ -492,7 +650,7 @@ class LoadPointCloud:
|
||||
}
|
||||
}
|
||||
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/PointCloud"
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("loaded pointcloud",)
|
||||
FUNCTION = "load_pointcloud"
|
||||
@@ -504,24 +662,45 @@ class LoadPointCloud:
|
||||
arr = np.load(file_path)
|
||||
tensor_pc = torch.from_numpy(arr)
|
||||
return (tensor_pc,)
|
||||
coords = []
|
||||
colors = []
|
||||
with open(file_path, 'r') as f:
|
||||
line = f.readline().strip()
|
||||
while not line.startswith("end_header"):
|
||||
|
||||
if o3d is None:
|
||||
logging.warning("[camera-comfyUI] open3d is not installed. Falling back to manual PLY parser.")
|
||||
coords = []
|
||||
colors = []
|
||||
with open(file_path, 'r') as f:
|
||||
line = f.readline().strip()
|
||||
for line in f:
|
||||
parts = line.strip().split()
|
||||
if len(parts) < 7:
|
||||
continue
|
||||
x, y, z = map(float, parts[0:3])
|
||||
r, g, b, a = map(int, parts[3:7])
|
||||
coords.append((x, y, z))
|
||||
colors.append((r, g, b, a))
|
||||
np_coords = np.array(coords, dtype=np.float32)
|
||||
np_colors = np.array(colors, dtype=np.float32)/255.0
|
||||
combined = np.concatenate([np_coords, np_colors], axis=1)
|
||||
tensor_pc = torch.from_numpy(combined)
|
||||
while not line.startswith("end_header"):
|
||||
line = f.readline().strip()
|
||||
for line in f:
|
||||
parts = line.strip().split()
|
||||
if len(parts) < 7:
|
||||
continue
|
||||
x, y, z = map(float, parts[0:3])
|
||||
r, g, b, a = map(float, parts[3:7])
|
||||
coords.append((x, y, z))
|
||||
colors.append((r, g, b, a))
|
||||
np_coords = np.array(coords, dtype=np.float32)
|
||||
np_colors = np.array(colors, dtype=np.float32)
|
||||
# if colors are > 1, normalize them to [0,1]
|
||||
if np_colors.max() > 1.0:
|
||||
np_colors = np_colors / 255.0
|
||||
else:
|
||||
pc = o3d.t.io.read_point_cloud(file_path)
|
||||
np_coords = pc.point["positions"].numpy().astype(np.float32)
|
||||
if "colors" in pc.point:
|
||||
cols = pc.point["colors"].numpy().astype(np.float32)
|
||||
else:
|
||||
cols = np.ones((np_coords.shape[0], 3), dtype=np.float32)
|
||||
if "alpha" in pc.point:
|
||||
alpha = pc.point["alpha"].numpy().astype(np.float32)
|
||||
else:
|
||||
alpha = np.ones((np_coords.shape[0], 1), dtype=np.float32)
|
||||
np_colors = np.concatenate([cols, alpha], axis=1)
|
||||
if np_colors.max() > 1.0:
|
||||
np_colors = np_colors / 255.0
|
||||
# combine coords and colors into a single tensor
|
||||
combined = np.concatenate([np_coords, np_colors], axis=1)
|
||||
tensor_pc = torch.from_numpy(combined)
|
||||
return (tensor_pc,)
|
||||
|
||||
@classmethod
|
||||
@@ -569,7 +748,7 @@ class SavePointCloud:
|
||||
RETURN_TYPES = ()
|
||||
FUNCTION = "save_pointcloud"
|
||||
OUTPUT_NODE = True
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/PointCloud"
|
||||
DESCRIPTION = "Saves the input point cloud to your ComfyUI output directory as .ply or .npy."
|
||||
|
||||
def save_pointcloud(self, pointcloud: torch.Tensor, filename_prefix: str, save_as: str = "ply"):
|
||||
@@ -586,24 +765,36 @@ class SavePointCloud:
|
||||
os.makedirs(full_output_folder, exist_ok=True)
|
||||
base_name = filename.replace("%batch_num%", "0")
|
||||
if save_as == "ply":
|
||||
ply_name = f"{base_name}_{counter:05}.ply"
|
||||
ply_path = os.path.join(full_output_folder, ply_name)
|
||||
coords = pointcloud[:, :3].cpu().numpy()
|
||||
colors = pointcloud[:, 3:].cpu().numpy().clip(0,1)
|
||||
with open(ply_path, 'w') as f:
|
||||
f.write("ply\n")
|
||||
f.write("format ascii 1.0\n")
|
||||
f.write(f"element vertex {coords.shape[0]}\n")
|
||||
f.write("property float x\n")
|
||||
f.write("property float y\n")
|
||||
f.write("property float z\n")
|
||||
f.write("property uchar red\n")
|
||||
f.write("property uchar green\n")
|
||||
f.write("property uchar blue\n")
|
||||
f.write("property uchar alpha\n")
|
||||
f.write("end_header\n")
|
||||
for (x,y,z), (r,g,b,a) in zip(coords, colors):
|
||||
f.write(f"{x} {y} {z} {int(r*255)} {int(g*255)} {int(b*255)} {int(a*255)}\n")
|
||||
ply_name = f"{base_name}_{counter:05}.ply"
|
||||
ply_path = os.path.join(full_output_folder, ply_name)
|
||||
coords = pointcloud[:, :3].cpu().numpy().astype(np.float32)
|
||||
colors = pointcloud[:, 3:].cpu().numpy().clip(0, 1).astype(np.float32)
|
||||
|
||||
if o3d is None:
|
||||
logging.warning("[camera-comfyUI] open3d is not installed. Falling back to manual ASCII PLY writer.")
|
||||
with open(ply_path, 'w') as f:
|
||||
f.write("ply\n")
|
||||
f.write("format ascii 1.0\n")
|
||||
f.write(f"element vertex {coords.shape[0]}\n")
|
||||
f.write("property float x\n")
|
||||
f.write("property float y\n")
|
||||
f.write("property float z\n")
|
||||
f.write("property float red\n")
|
||||
f.write("property float green\n")
|
||||
f.write("property float blue\n")
|
||||
f.write("property float alpha\n")
|
||||
f.write("end_header\n")
|
||||
for (x, y, z), (r, g, b, a) in zip(coords, colors):
|
||||
f.write(f"{x} {y} {z} {r} {g} {b} {a}\n")
|
||||
else:
|
||||
pc = o3d.t.geometry.PointCloud()
|
||||
pc.point["positions"] = o3d.core.Tensor(coords, o3d.core.float32)
|
||||
pc.point["colors"] = o3d.core.Tensor(colors[:, :3], o3d.core.float32)
|
||||
if colors.shape[1] > 3:
|
||||
pc.point["alpha"] = o3d.core.Tensor(colors[:, 3:], o3d.core.float32)
|
||||
else:
|
||||
pc.point["alpha"] = o3d.core.Tensor(np.ones((coords.shape[0], 1), dtype=np.float32), o3d.core.float32)
|
||||
o3d.t.io.write_point_cloud(ply_path, pc)
|
||||
file_name = ply_name
|
||||
else:
|
||||
npy_name = f"{base_name}_{counter:05}.npy"
|
||||
@@ -640,12 +831,15 @@ class CameraMotionNode:
|
||||
"output_width": ("INT", {"default":512, "min":8, "max":16384}),
|
||||
"output_height": ("INT", {"default":512, "min":8, "max":16384}),
|
||||
"point_size": ("INT", {"default":1, "min":1}),
|
||||
"widen_mask": ("INT", {"default":0, "min":0, "max":64}),
|
||||
"invert_mask": ("BOOLEAN", {"default": False}),
|
||||
"points_to_mask": ("BOOLEAN", {"default": False, "tooltip": "Output mask frames of projected points"}),
|
||||
}}
|
||||
|
||||
RETURN_TYPES = ("IMAGE",)
|
||||
RETURN_NAMES = ("motion_frames",)
|
||||
RETURN_TYPES = ("IMAGE", "MASK")
|
||||
RETURN_NAMES = ("motion_frames", "mask_frames")
|
||||
FUNCTION = "generate_motion_frames"
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/Trajectory"
|
||||
|
||||
def generate_motion_frames(
|
||||
self,
|
||||
@@ -656,7 +850,10 @@ class CameraMotionNode:
|
||||
output_horizontal_fov: float,
|
||||
output_width: int,
|
||||
output_height: int,
|
||||
point_size: int = 1
|
||||
point_size: int = 1,
|
||||
widen_mask: int = 0,
|
||||
invert_mask: bool = False,
|
||||
points_to_mask: bool = False
|
||||
) -> Tuple[torch.Tensor]:
|
||||
# validate trajectory shape
|
||||
if trajectory.dim() != 3 or trajectory.shape[1:] != (4,4):
|
||||
@@ -683,9 +880,10 @@ class CameraMotionNode:
|
||||
proj_node = ProjectPointCloud()
|
||||
transform_node = TransformPointCloud()
|
||||
frames = []
|
||||
masks = []
|
||||
for M in tqdm(full_traj):
|
||||
pc_t, = transform_node.transform_pointcloud(pointcloud, M)
|
||||
img, _, _ = proj_node.project_pointcloud(
|
||||
img, mask, _ = proj_node.project_pointcloud(
|
||||
pc_t,
|
||||
output_projection,
|
||||
output_horizontal_fov,
|
||||
@@ -693,15 +891,25 @@ class CameraMotionNode:
|
||||
output_height,
|
||||
point_size
|
||||
)
|
||||
if widen_mask > 0:
|
||||
k = 2 * widen_mask + 1
|
||||
pad = widen_mask
|
||||
mask = F.max_pool2d(mask.float().unsqueeze(0).unsqueeze(0), kernel_size=k, stride=1, padding=pad).squeeze(0).squeeze(0)
|
||||
if invert_mask:
|
||||
mask = 1.0 - mask
|
||||
masks.append(mask)
|
||||
if points_to_mask:
|
||||
img = mask.unsqueeze(-1).repeat(1,1,1,3)
|
||||
frames.append(img[0])
|
||||
|
||||
# output as (T,H,W,3)
|
||||
return (torch.stack(frames, dim=0),)
|
||||
return (torch.stack(frames, dim=0), torch.stack(masks, dim=0))
|
||||
|
||||
class CameraInterpolationNode:
|
||||
"""
|
||||
Wrap two 4×4 poses into a trajectory tensor.
|
||||
Outputs only `trajectory` (shape 2×4×4).
|
||||
Interpolate between two 4×4 poses into a trajectory tensor using proper
|
||||
SE(3) interpolation (quaternion SLERP on rotation, lerp on translation).
|
||||
Outputs `trajectory` (shape num_steps×4×4, default 2×4×4).
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
@@ -710,25 +918,29 @@ class CameraInterpolationNode:
|
||||
"required": {
|
||||
"initial_matrix": ("MAT_4X4",),
|
||||
"final_matrix": ("MAT_4X4",),
|
||||
}
|
||||
},
|
||||
"optional": {
|
||||
"num_steps": ("INT", {"default": 2, "min": 2, "max": 4096, "tooltip": "Number of poses in the output trajectory, SE(3)-interpolated between the two matrices."}),
|
||||
},
|
||||
}
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("trajectory",)
|
||||
FUNCTION = "interpolate"
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/Trajectory"
|
||||
|
||||
def interpolate(
|
||||
self,
|
||||
initial_matrix: torch.Tensor,
|
||||
final_matrix: torch.Tensor,
|
||||
num_steps: int = 2,
|
||||
) -> Tuple[torch.Tensor]:
|
||||
# stack into a (2,4,4) trajectory
|
||||
# convert to tensor if needed
|
||||
if isinstance(initial_matrix, np.ndarray):
|
||||
initial_matrix = torch.from_numpy(initial_matrix).float()
|
||||
if isinstance(final_matrix, np.ndarray):
|
||||
final_matrix = torch.from_numpy(final_matrix).float()
|
||||
traj = torch.stack([initial_matrix, final_matrix], dim=0)
|
||||
keyframes = torch.stack([initial_matrix.float(), final_matrix.float()], dim=0)
|
||||
traj = interpolate_se3(keyframes, num_steps)
|
||||
return (traj,)
|
||||
|
||||
|
||||
@@ -743,14 +955,14 @@ class CameraTrajectoryNode:
|
||||
"pointcloud": ("TENSOR",),
|
||||
},
|
||||
"optional": {
|
||||
"initial_matrix": ("MAT_4X4"),
|
||||
"initial_matrix": ("MAT_4X4",),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("trajectory",)
|
||||
FUNCTION = "build_trajectory"
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/Trajectory"
|
||||
|
||||
def build_trajectory(
|
||||
self,
|
||||
@@ -890,7 +1102,7 @@ class PointCloudCleaner:
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("cleaned_pointcloud",)
|
||||
FUNCTION = "clean_pointcloud"
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/PointCloud"
|
||||
|
||||
def clean_pointcloud(
|
||||
self,
|
||||
@@ -966,7 +1178,7 @@ class ProjectAndClean:
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("cleaned_pointcloud",)
|
||||
FUNCTION = "project_and_clean"
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/PointCloud"
|
||||
|
||||
def project_and_clean(
|
||||
self,
|
||||
@@ -1080,7 +1292,7 @@ class SaveTrajectory:
|
||||
RETURN_TYPES = ()
|
||||
FUNCTION = "save_trajectory"
|
||||
OUTPUT_NODE = True
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/Trajectory"
|
||||
DESCRIPTION = "Saves the input trajectory tensor (N,4,4) to your ComfyUI output directory as .npy."
|
||||
|
||||
def save_trajectory(self, trajectory: torch.Tensor, filename_prefix: str):
|
||||
@@ -1130,7 +1342,7 @@ class LoadTrajectory:
|
||||
}
|
||||
}
|
||||
|
||||
CATEGORY = "Camera/pointcloud"
|
||||
CATEGORY = "Camera/Trajectory"
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("loaded_trajectory",)
|
||||
FUNCTION = "load_trajectory"
|
||||
@@ -1156,6 +1368,95 @@ class LoadTrajectory:
|
||||
return f"Invalid trajectory file: {trajectory_file}"
|
||||
return True
|
||||
|
||||
class DepthEdgeFilter:
|
||||
"""
|
||||
Detect "flying pixel" depth discontinuities and output a validity mask.
|
||||
A pixel is flagged as an edge where |depth gradient| / depth exceeds
|
||||
`relative_threshold`; edges are optionally dilated. Returns a MASK with
|
||||
1.0 where the depth is valid (NOT a flying-pixel edge) and 0.0 on edges.
|
||||
"""
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
# Depth: [H,W] or [T,H,W], trailing channel dim of 1 accepted
|
||||
"depth": ("TENSOR", {"shape_hint": [None, None, None]}),
|
||||
"relative_threshold": ("FLOAT", {"default": 0.05, "min": 0.0, "max": 10.0, "step": 0.005, "tooltip": "Mark a pixel as edge where |depth gradient| / depth exceeds this value."}),
|
||||
"dilate": ("INT", {"default": 1, "min": 0, "max": 64, "tooltip": "Grow detected edges by this many pixels (max-pool dilation)."}),
|
||||
},
|
||||
"optional": {
|
||||
"mask": ("MASK", {"tooltip": "Optional validity mask ANDed with the edge-filter result."}),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("MASK",)
|
||||
RETURN_NAMES = ("valid_mask",)
|
||||
FUNCTION = "filter_edges"
|
||||
CATEGORY = "Camera/PointCloud"
|
||||
|
||||
def filter_edges(
|
||||
self,
|
||||
depth: torch.Tensor,
|
||||
relative_threshold: float,
|
||||
dilate: int,
|
||||
mask: torch.Tensor = None,
|
||||
) -> Tuple[torch.Tensor]:
|
||||
d = depth
|
||||
if isinstance(d, np.ndarray):
|
||||
d = torch.from_numpy(d)
|
||||
d = d.float()
|
||||
# Accept [H,W], [H,W,1], [T,H,W], [T,H,W,1]
|
||||
if d.dim() == 4 and d.shape[-1] == 1:
|
||||
d = d[..., 0]
|
||||
elif d.dim() == 3 and d.shape[-1] == 1:
|
||||
d = d[..., 0]
|
||||
squeeze_batch = False
|
||||
if d.dim() == 2:
|
||||
d = d.unsqueeze(0)
|
||||
squeeze_batch = True
|
||||
if d.dim() != 3:
|
||||
raise ValueError(f"DepthEdgeFilter expects depth of shape [H,W] or [T,H,W] (trailing 1 ok), got {tuple(depth.shape)}")
|
||||
|
||||
eps = 1e-8
|
||||
# Forward differences along x and y; propagate each difference to both
|
||||
# neighbouring pixels so both sides of a discontinuity are flagged.
|
||||
dx = (d[:, :, 1:] - d[:, :, :-1]).abs()
|
||||
dy = (d[:, 1:, :] - d[:, :-1, :]).abs()
|
||||
gx = torch.zeros_like(d)
|
||||
gx[:, :, :-1] = dx
|
||||
gx[:, :, 1:] = torch.maximum(gx[:, :, 1:], dx)
|
||||
gy = torch.zeros_like(d)
|
||||
gy[:, :-1, :] = dy
|
||||
gy[:, 1:, :] = torch.maximum(gy[:, 1:, :], dy)
|
||||
grad = torch.maximum(gx, gy)
|
||||
edge = (grad / d.abs().clamp(min=eps)) > relative_threshold
|
||||
|
||||
if dilate > 0:
|
||||
k = 2 * int(dilate) + 1
|
||||
edge = F.max_pool2d(edge.float().unsqueeze(1), kernel_size=k, stride=1, padding=int(dilate)).squeeze(1) > 0.5
|
||||
|
||||
valid = (~edge).float()
|
||||
|
||||
if mask is not None:
|
||||
m = mask
|
||||
if isinstance(m, np.ndarray):
|
||||
m = torch.from_numpy(m)
|
||||
m = m.float().to(valid.device)
|
||||
if m.dim() == 4 and m.shape[-1] == 1:
|
||||
m = m[..., 0]
|
||||
if m.dim() == 2:
|
||||
m = m.unsqueeze(0)
|
||||
if m.shape[0] == 1 and valid.shape[0] > 1:
|
||||
m = m.expand(valid.shape[0], -1, -1)
|
||||
if m.shape[-2:] != valid.shape[-2:]:
|
||||
m = F.interpolate(m.unsqueeze(1), size=valid.shape[-2:], mode="nearest").squeeze(1)
|
||||
valid = valid * (m > 0.5).float()
|
||||
|
||||
if squeeze_batch:
|
||||
valid = valid[0]
|
||||
return (valid,)
|
||||
|
||||
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"DepthToPointCloud": DepthToPointCloud,
|
||||
"TransformPointCloud": TransformPointCloud,
|
||||
@@ -1170,4 +1471,5 @@ NODE_CLASS_MAPPINGS = {
|
||||
"PointCloudCleaner": PointCloudCleaner,
|
||||
"SaveTrajectory": SaveTrajectory,
|
||||
"LoadTrajectory": LoadTrajectory,
|
||||
"DepthEdgeFilter": DepthEdgeFilter,
|
||||
}
|
||||
@@ -0,0 +1,462 @@
|
||||
"""Camera pose estimation nodes.
|
||||
|
||||
Provides:
|
||||
- VideoPoseEstimator: VGGT-based per-frame camera pose + depth + intrinsics
|
||||
estimation from a video clip.
|
||||
- TrajectoryInvert / TrajectoryCompose: small utility nodes for wiring
|
||||
trajectory tensors ([K, 4, 4] world-to-camera matrices) in graphs.
|
||||
|
||||
Coordinate convention (matches the rest of this repo): camera frame is
|
||||
+X right, +Y down, +Z forward; trajectory matrices are 4x4 world-to-camera
|
||||
(`cam_pts = world_pts @ R.T + t`). VGGT outputs OpenCV-convention
|
||||
camera-from-world extrinsics, which match this convention directly.
|
||||
"""
|
||||
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
from tqdm import tqdm
|
||||
|
||||
try:
|
||||
import folder_paths
|
||||
except ImportError: # Allow notebook usage outside ComfyUI
|
||||
class _FolderPathsStub:
|
||||
def __getattr__(self, name):
|
||||
raise ModuleNotFoundError(
|
||||
"folder_paths is unavailable; this feature requires the ComfyUI runtime."
|
||||
)
|
||||
|
||||
folder_paths = _FolderPathsStub()
|
||||
|
||||
_here = os.path.dirname(os.path.abspath(__file__))
|
||||
# climb up 2 levels: camera-comfyUI -> custom_nodes -> ComfyUI
|
||||
COMFYUI_ROOT = os.path.abspath(os.path.join(_here, os.pardir, os.pardir))
|
||||
|
||||
DEVICE_CHOICES = ["auto", "cpu", "cuda"]
|
||||
|
||||
# Module-level model cache: {device_str: model}
|
||||
_VGGT_MODEL_CACHE: Dict[str, Any] = {}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# SE(3) interpolation (contract C1). Prefer the shared implementation from
|
||||
# pointcloud_nodes; fall back to a local copy so this file works standalone.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def _matrix_to_quaternion(R: torch.Tensor) -> torch.Tensor:
|
||||
"""Convert a single 3x3 rotation matrix to a wxyz quaternion."""
|
||||
R = R.to(torch.float64)
|
||||
m00, m01, m02 = R[0, 0], R[0, 1], R[0, 2]
|
||||
m10, m11, m12 = R[1, 0], R[1, 1], R[1, 2]
|
||||
m20, m21, m22 = R[2, 0], R[2, 1], R[2, 2]
|
||||
trace = m00 + m11 + m22
|
||||
if trace > 0.0:
|
||||
s = torch.sqrt(trace + 1.0) * 2.0
|
||||
w = 0.25 * s
|
||||
x = (m21 - m12) / s
|
||||
y = (m02 - m20) / s
|
||||
z = (m10 - m01) / s
|
||||
elif (m00 > m11) and (m00 > m22):
|
||||
s = torch.sqrt(1.0 + m00 - m11 - m22) * 2.0
|
||||
w = (m21 - m12) / s
|
||||
x = 0.25 * s
|
||||
y = (m01 + m10) / s
|
||||
z = (m02 + m20) / s
|
||||
elif m11 > m22:
|
||||
s = torch.sqrt(1.0 + m11 - m00 - m22) * 2.0
|
||||
w = (m02 - m20) / s
|
||||
x = (m01 + m10) / s
|
||||
y = 0.25 * s
|
||||
z = (m12 + m21) / s
|
||||
else:
|
||||
s = torch.sqrt(1.0 + m22 - m00 - m11) * 2.0
|
||||
w = (m10 - m01) / s
|
||||
x = (m02 + m20) / s
|
||||
y = (m12 + m21) / s
|
||||
z = 0.25 * s
|
||||
q = torch.stack([w, x, y, z])
|
||||
return (q / q.norm().clamp(min=1e-12)).to(torch.float32)
|
||||
|
||||
|
||||
def _quaternion_to_matrix(q: torch.Tensor) -> torch.Tensor:
|
||||
"""Convert a wxyz quaternion to a 3x3 rotation matrix."""
|
||||
q = q / q.norm().clamp(min=1e-12)
|
||||
w, x, y, z = q[0], q[1], q[2], q[3]
|
||||
return torch.stack([
|
||||
torch.stack([1 - 2 * (y * y + z * z), 2 * (x * y - w * z), 2 * (x * z + w * y)]),
|
||||
torch.stack([2 * (x * y + w * z), 1 - 2 * (x * x + z * z), 2 * (y * z - w * x)]),
|
||||
torch.stack([2 * (x * z - w * y), 2 * (y * z + w * x), 1 - 2 * (x * x + y * y)]),
|
||||
])
|
||||
|
||||
|
||||
def _quat_slerp(q0: torch.Tensor, q1: torch.Tensor, alpha: float) -> torch.Tensor:
|
||||
"""Spherical linear interpolation between two wxyz quaternions."""
|
||||
q0 = q0 / q0.norm().clamp(min=1e-12)
|
||||
q1 = q1 / q1.norm().clamp(min=1e-12)
|
||||
dot = torch.dot(q0, q1)
|
||||
if dot < 0.0: # take the short path on the quaternion hypersphere
|
||||
q1 = -q1
|
||||
dot = -dot
|
||||
dot = dot.clamp(-1.0, 1.0)
|
||||
if dot > 0.9995: # nearly parallel: lerp + renormalize is numerically safer
|
||||
q = (1.0 - alpha) * q0 + alpha * q1
|
||||
return q / q.norm().clamp(min=1e-12)
|
||||
theta = torch.acos(dot)
|
||||
sin_theta = torch.sin(theta)
|
||||
w0 = torch.sin((1.0 - alpha) * theta) / sin_theta
|
||||
w1 = torch.sin(alpha * theta) / sin_theta
|
||||
q = w0 * q0 + w1 * q1
|
||||
return q / q.norm().clamp(min=1e-12)
|
||||
|
||||
|
||||
def _interpolate_se3_fallback(trajectory: torch.Tensor, num_steps: int) -> torch.Tensor:
|
||||
"""trajectory [K,4,4] -> [num_steps,4,4]. Piecewise: quaternion SLERP on R, lerp on t.
|
||||
K==1 -> repeat. Returns valid (orthonormal) rotation matrices. Matches contract C1."""
|
||||
traj = torch.as_tensor(trajectory, dtype=torch.float32)
|
||||
if traj.dim() == 2:
|
||||
traj = traj.unsqueeze(0)
|
||||
if traj.dim() != 3 or traj.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"Expected trajectory of shape [K,4,4], got {tuple(traj.shape)}")
|
||||
K = traj.shape[0]
|
||||
if K == 1:
|
||||
return traj.expand(num_steps, 4, 4).clone()
|
||||
quats = torch.stack([_matrix_to_quaternion(traj[i, :3, :3]) for i in range(K)])
|
||||
trans = traj[:, :3, 3]
|
||||
positions = torch.linspace(0.0, float(K - 1), num_steps)
|
||||
out = []
|
||||
for pos in positions:
|
||||
lower = int(torch.floor(pos).clamp(max=K - 2))
|
||||
upper = lower + 1
|
||||
alpha = float(pos) - lower
|
||||
q = _quat_slerp(quats[lower], quats[upper], alpha)
|
||||
t = (1.0 - alpha) * trans[lower] + alpha * trans[upper]
|
||||
M = torch.eye(4, dtype=torch.float32)
|
||||
M[:3, :3] = _quaternion_to_matrix(q)
|
||||
M[:3, 3] = t
|
||||
out.append(M)
|
||||
return torch.stack(out, dim=0)
|
||||
|
||||
|
||||
try:
|
||||
from .pointcloud_nodes import interpolate_se3
|
||||
except Exception:
|
||||
try:
|
||||
from pointcloud_nodes import interpolate_se3
|
||||
except Exception:
|
||||
interpolate_se3 = _interpolate_se3_fallback
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# VGGT lazy import helpers
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def _import_vggt() -> Tuple[Any, Any]:
|
||||
"""Lazily import VGGT. Tries the pip package first, then a sibling clone
|
||||
at COMFYUI_ROOT/vggt (mirroring how video_nodes.py handles Video-Depth-Anything)."""
|
||||
try:
|
||||
from vggt.models.vggt import VGGT
|
||||
from vggt.utils.pose_enc import pose_encoding_to_extri_intri
|
||||
return VGGT, pose_encoding_to_extri_intri
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
vggt_clone_path = os.path.join(COMFYUI_ROOT, "vggt")
|
||||
if os.path.isdir(vggt_clone_path) and vggt_clone_path not in sys.path:
|
||||
sys.path.insert(0, vggt_clone_path)
|
||||
try:
|
||||
from vggt.models.vggt import VGGT
|
||||
from vggt.utils.pose_enc import pose_encoding_to_extri_intri
|
||||
return VGGT, pose_encoding_to_extri_intri
|
||||
except ImportError as exc:
|
||||
raise ModuleNotFoundError(
|
||||
"VGGT is not installed. Run this pack's install.py (ComfyUI-Manager does this "
|
||||
"automatically), or install it manually with "
|
||||
"`pip install git+https://github.com/facebookresearch/vggt.git` (it is not on PyPI), "
|
||||
f"or clone https://github.com/facebookresearch/vggt into {vggt_clone_path!r}. "
|
||||
"It also requires `huggingface_hub` to download the facebook/VGGT-1B weights."
|
||||
) from exc
|
||||
|
||||
|
||||
def _get_vggt_model(device: torch.device) -> Any:
|
||||
"""Load (and cache) the VGGT-1B model on the requested device."""
|
||||
key = str(device)
|
||||
if key not in _VGGT_MODEL_CACHE:
|
||||
VGGT, _ = _import_vggt()
|
||||
print(f"[pose_nodes] Loading facebook/VGGT-1B onto {key} (first call downloads ~5GB weights)...")
|
||||
model = VGGT.from_pretrained("facebook/VGGT-1B")
|
||||
model = model.to(device).eval()
|
||||
_VGGT_MODEL_CACHE[key] = model
|
||||
return _VGGT_MODEL_CACHE[key]
|
||||
|
||||
|
||||
def _vggt_preprocess(frames: torch.Tensor, resolution: int, device: torch.device) -> torch.Tensor:
|
||||
"""[T,H,W,3] float 0..1 -> [1,T,3,Hp,Wp] with max dim == resolution (both dims
|
||||
divisible by 14, the VGGT patch size), aspect ratio preserved."""
|
||||
T, H, W, _ = frames.shape
|
||||
imgs = frames.permute(0, 3, 1, 2).to(device=device, dtype=torch.float32)
|
||||
if imgs.max() > 1.5: # defensively handle 0..255 inputs
|
||||
imgs = imgs / 255.0
|
||||
scale = float(resolution) / float(max(H, W))
|
||||
new_h = max(14, int(round(H * scale / 14.0)) * 14)
|
||||
new_w = max(14, int(round(W * scale / 14.0)) * 14)
|
||||
if (new_h, new_w) != (H, W):
|
||||
imgs = F.interpolate(imgs, size=(new_h, new_w), mode="bilinear", align_corners=False)
|
||||
return imgs.clamp(0.0, 1.0).unsqueeze(0) # [1,T,3,Hp,Wp]
|
||||
|
||||
|
||||
class VideoPoseEstimator:
|
||||
"""
|
||||
Estimates per-frame camera poses (world-to-camera [T,4,4]), metric-ish depth
|
||||
maps, depth confidence and the horizontal FOV from a video clip using
|
||||
facebook/VGGT-1B.
|
||||
|
||||
VGGT extrinsics use the OpenCV camera convention (+X right, +Y down,
|
||||
+Z forward, camera-from-world), which matches this repo's trajectory
|
||||
convention, so the matrices are returned as-is (padded to 4x4).
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
# Video frames: Tensor [T, H, W, 3] float 0..1
|
||||
"frames": ("IMAGE", {"shape_hint": [None, None, None, 3]}),
|
||||
"max_frames": ("INT", {
|
||||
"default": 64, "min": 1, "max": 1024,
|
||||
"tooltip": "If the clip has more frames than this, it is stride-subsampled "
|
||||
"for VGGT and the poses are SE(3)-interpolated back to full length "
|
||||
"(depth/confidence use nearest-frame fill).",
|
||||
}),
|
||||
"resolution": ("INT", {
|
||||
"default": 518, "min": 98, "max": 1036,
|
||||
"tooltip": "Max image dimension fed to VGGT (rounded to a multiple of 14).",
|
||||
}),
|
||||
"device": (DEVICE_CHOICES, {"default": "auto"}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("TENSOR", "TENSOR", "FLOAT", "TENSOR")
|
||||
RETURN_NAMES = ("trajectory", "depths", "horizontal_fov", "confidence")
|
||||
FUNCTION = "estimate_poses"
|
||||
CATEGORY = "Camera/Pose"
|
||||
DESCRIPTION = (
|
||||
"VGGT camera pose + depth estimation. Outputs world-to-camera trajectory [T,4,4], "
|
||||
"depth maps [T,H,W] at the input resolution, mean horizontal FOV (degrees) and "
|
||||
"per-pixel depth confidence [T,H,W]."
|
||||
)
|
||||
|
||||
def estimate_poses(
|
||||
self,
|
||||
frames: torch.Tensor,
|
||||
max_frames: int = 64,
|
||||
resolution: int = 518,
|
||||
device: str = "auto",
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, float, torch.Tensor]:
|
||||
if frames.dim() != 4 or frames.shape[-1] != 3:
|
||||
raise ValueError(f"Expected frames of shape [T,H,W,3], got {tuple(frames.shape)}")
|
||||
T_full, H, W, _ = frames.shape
|
||||
|
||||
if device == "auto":
|
||||
dev = torch.device("cuda" if torch.cuda.is_available() else "cpu")
|
||||
elif device == "cuda":
|
||||
if not torch.cuda.is_available():
|
||||
raise ValueError("CUDA requested but not available.")
|
||||
dev = torch.device("cuda")
|
||||
else:
|
||||
dev = torch.device("cpu")
|
||||
|
||||
# Stride-subsample overly long clips, keeping the frame mapping so that
|
||||
# poses can be interpolated back afterwards.
|
||||
if T_full > max_frames:
|
||||
sub_indices = torch.linspace(0, T_full - 1, max_frames).round().long().unique()
|
||||
print(
|
||||
f"[VideoPoseEstimator] WARNING: clip has {T_full} frames > max_frames={max_frames}; "
|
||||
f"running VGGT on {sub_indices.numel()} stride-subsampled frames. Poses are "
|
||||
"SE(3)-interpolated back to full length; depth/confidence use nearest-frame fill. "
|
||||
"Increase max_frames for exact per-frame estimates."
|
||||
)
|
||||
proc_frames = frames[sub_indices]
|
||||
else:
|
||||
sub_indices = None
|
||||
proc_frames = frames
|
||||
|
||||
images = _vggt_preprocess(proc_frames, resolution, dev) # [1,S,3,Hp,Wp]
|
||||
S, Hp, Wp = images.shape[1], images.shape[-2], images.shape[-1]
|
||||
|
||||
_, pose_encoding_to_extri_intri = _import_vggt()
|
||||
model = _get_vggt_model(dev)
|
||||
|
||||
try:
|
||||
with torch.no_grad():
|
||||
if dev.type == "cuda":
|
||||
capability = torch.cuda.get_device_capability(dev)
|
||||
amp_dtype = torch.bfloat16 if capability[0] >= 8 else torch.float16
|
||||
with torch.autocast(device_type="cuda", dtype=amp_dtype):
|
||||
aggregated_tokens_list, ps_idx = model.aggregator(images)
|
||||
else:
|
||||
aggregated_tokens_list, ps_idx = model.aggregator(images)
|
||||
# Camera + depth heads run in full precision (per the official VGGT example).
|
||||
pose_enc = model.camera_head(aggregated_tokens_list)[-1]
|
||||
extrinsic, intrinsic = pose_encoding_to_extri_intri(pose_enc, images.shape[-2:])
|
||||
depth_map, depth_conf = model.depth_head(aggregated_tokens_list, images, ps_idx)
|
||||
except torch.cuda.OutOfMemoryError as exc:
|
||||
raise RuntimeError(
|
||||
f"VGGT ran out of GPU memory on {S} frames at {Wp}x{Hp}. "
|
||||
"Lower max_frames and/or resolution, or set device='cpu' (slow)."
|
||||
) from exc
|
||||
|
||||
# ---- Trajectory: pad OpenCV world-to-camera [S,3,4] to [S,4,4] ---- #
|
||||
extrinsic = extrinsic.squeeze(0).to(torch.float32).cpu() # [S,3,4]
|
||||
trajectory = torch.eye(4, dtype=torch.float32).unsqueeze(0).repeat(extrinsic.shape[0], 1, 1)
|
||||
trajectory[:, :3, :4] = extrinsic
|
||||
|
||||
# ---- Horizontal FOV from intrinsics (resolution-invariant fx/W ratio) ---- #
|
||||
intrinsic = intrinsic.squeeze(0).to(torch.float32).cpu() # [S,3,3]
|
||||
fx = intrinsic[:, 0, 0].clamp(min=1e-6)
|
||||
hfov_per_frame = 2.0 * torch.atan(0.5 * float(Wp) / fx) # radians, at processing width
|
||||
# Aspect ratio is preserved during preprocessing, so fx/W is the same at
|
||||
# the original width and the FOV needs no conversion.
|
||||
horizontal_fov = float(torch.rad2deg(hfov_per_frame).mean())
|
||||
|
||||
# ---- Depth + confidence, resized back to the input resolution ---- #
|
||||
depth = depth_map.squeeze(0).to(torch.float32).cpu() # [S,Hp,Wp,1] (or [S,Hp,Wp])
|
||||
if depth.dim() == 4 and depth.shape[-1] == 1:
|
||||
depth = depth.squeeze(-1)
|
||||
conf = depth_conf.squeeze(0).to(torch.float32).cpu() # [S,Hp,Wp]
|
||||
if conf.dim() == 4 and conf.shape[-1] == 1:
|
||||
conf = conf.squeeze(-1)
|
||||
|
||||
# ---- Convert VGGT z-depth to RADIAL ray depth ---- #
|
||||
# VGGT's depth head predicts z-depth (its unprojection is
|
||||
# x = (u - cx) * d / fx, z = d), while every consumer in this repo
|
||||
# (pointcloud *_depth_to_XYZ helpers, MotionMaskFromDepth,
|
||||
# TracksToTrajectories, the GS4D helpers) multiplies unit ray directions
|
||||
# by depth, i.e. expects RADIAL distance. Multiply by the per-pixel ray
|
||||
# norm sqrt(1 + ((u-cx)/fx)^2 + ((v-cy)/fy)^2) using the per-frame
|
||||
# intrinsics at the VGGT processing resolution.
|
||||
fx_pf = intrinsic[:, 0, 0].clamp(min=1e-6).view(-1, 1, 1) # [S,1,1]
|
||||
fy_pf = intrinsic[:, 1, 1].clamp(min=1e-6).view(-1, 1, 1)
|
||||
cx_pf = intrinsic[:, 0, 2].view(-1, 1, 1)
|
||||
cy_pf = intrinsic[:, 1, 2].view(-1, 1, 1)
|
||||
uu = torch.arange(Wp, dtype=torch.float32).view(1, 1, -1)
|
||||
vv = torch.arange(Hp, dtype=torch.float32).view(1, -1, 1)
|
||||
xn = (uu - cx_pf) / fx_pf
|
||||
yn = (vv - cy_pf) / fy_pf
|
||||
depth = depth * torch.sqrt(1.0 + xn * xn + yn * yn)
|
||||
|
||||
if (Hp, Wp) != (H, W):
|
||||
depth = F.interpolate(depth.unsqueeze(1), size=(H, W), mode="bilinear", align_corners=False).squeeze(1)
|
||||
conf = F.interpolate(conf.unsqueeze(1), size=(H, W), mode="bilinear", align_corners=False).squeeze(1)
|
||||
|
||||
# ---- If subsampled, expand back to the full frame count ---- #
|
||||
if sub_indices is not None:
|
||||
# Subsample indices are (near-)uniform over [0, T_full-1], so uniform
|
||||
# SE(3) resampling reconstructs per-frame poses well.
|
||||
trajectory = interpolate_se3(trajectory, T_full)
|
||||
all_t = torch.arange(T_full).unsqueeze(1) # [T_full,1]
|
||||
nearest = (sub_indices.unsqueeze(0) - all_t).abs().argmin(dim=1) # [T_full]
|
||||
depth = depth[nearest]
|
||||
conf = conf[nearest]
|
||||
|
||||
return (trajectory, depth, horizontal_fov, conf)
|
||||
|
||||
|
||||
class TrajectoryInvert:
|
||||
"""
|
||||
Inverts each 4x4 matrix in a trajectory tensor, converting between
|
||||
world-to-camera and camera-to-world conventions.
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
# Trajectory: Tensor [K, 4, 4] (a single [4, 4] matrix also works)
|
||||
"trajectory": ("TENSOR", {"shape_hint": [None, 4, 4]}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("trajectory",)
|
||||
FUNCTION = "invert"
|
||||
CATEGORY = "Camera/Pose"
|
||||
DESCRIPTION = "Inverts each 4x4 pose (world-to-camera <-> camera-to-world)."
|
||||
|
||||
def invert(self, trajectory: torch.Tensor) -> Tuple[torch.Tensor]:
|
||||
traj = torch.as_tensor(trajectory, dtype=torch.float32)
|
||||
squeeze = traj.dim() == 2
|
||||
if squeeze:
|
||||
traj = traj.unsqueeze(0)
|
||||
if traj.dim() != 3 or traj.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"Expected trajectory of shape [K,4,4], got {tuple(trajectory.shape)}")
|
||||
# Rigid-body inverse: R -> R.T, t -> -R.T @ t (numerically stabler than
|
||||
# a generic matrix inverse for SE(3) poses).
|
||||
R = traj[:, :3, :3]
|
||||
t = traj[:, :3, 3:4]
|
||||
Rt = R.transpose(1, 2)
|
||||
inv = torch.eye(4, dtype=traj.dtype).unsqueeze(0).repeat(traj.shape[0], 1, 1)
|
||||
inv[:, :3, :3] = Rt
|
||||
inv[:, :3, 3:4] = -Rt @ t
|
||||
if squeeze:
|
||||
inv = inv.squeeze(0)
|
||||
return (inv,)
|
||||
|
||||
|
||||
class TrajectoryCompose:
|
||||
"""
|
||||
Composes two trajectories per frame: out_k = A_k @ B_k. Either input may be
|
||||
a single [4,4] matrix, which is broadcast against the other. Useful for
|
||||
retargeting novel camera paths relative to a source pose (e.g. compose a
|
||||
relative path with the inverse of source pose 0).
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
# Left operand: Tensor [K, 4, 4] or [4, 4]
|
||||
"trajectory_a": ("TENSOR", {"shape_hint": [None, 4, 4]}),
|
||||
# Right operand: Tensor [K, 4, 4] or [4, 4]
|
||||
"trajectory_b": ("TENSOR", {"shape_hint": [None, 4, 4]}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("trajectory",)
|
||||
FUNCTION = "compose"
|
||||
CATEGORY = "Camera/Pose"
|
||||
DESCRIPTION = "Per-frame matrix product A @ B; a single 4x4 input broadcasts over the other."
|
||||
|
||||
def compose(self, trajectory_a: torch.Tensor, trajectory_b: torch.Tensor) -> Tuple[torch.Tensor]:
|
||||
A = torch.as_tensor(trajectory_a, dtype=torch.float32)
|
||||
B = torch.as_tensor(trajectory_b, dtype=torch.float32)
|
||||
both_single = A.dim() == 2 and B.dim() == 2
|
||||
if A.dim() == 2:
|
||||
A = A.unsqueeze(0)
|
||||
if B.dim() == 2:
|
||||
B = B.unsqueeze(0)
|
||||
if A.dim() != 3 or A.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"Expected trajectory_a of shape [K,4,4] or [4,4], got {tuple(trajectory_a.shape)}")
|
||||
if B.dim() != 3 or B.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"Expected trajectory_b of shape [K,4,4] or [4,4], got {tuple(trajectory_b.shape)}")
|
||||
if A.shape[0] != B.shape[0] and A.shape[0] != 1 and B.shape[0] != 1:
|
||||
raise ValueError(
|
||||
f"Trajectory lengths do not broadcast: {A.shape[0]} vs {B.shape[0]} "
|
||||
"(they must match, or one must be a single 4x4 matrix)."
|
||||
)
|
||||
out = torch.matmul(A, B) # broadcasts [1,4,4] against [K,4,4]
|
||||
if both_single:
|
||||
out = out.squeeze(0)
|
||||
return (out,)
|
||||
|
||||
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"VideoPoseEstimator": VideoPoseEstimator,
|
||||
"TrajectoryInvert": TrajectoryInvert,
|
||||
"TrajectoryCompose": TrajectoryCompose,
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
[project]
|
||||
name = "camera-comfyui"
|
||||
description = "Custom ComfyUI nodes for camera projections (pinhole/fisheye/equirectangular), depth, point clouds, camera trajectories, and 3D/4D Gaussian splatting — including video-to-4D-world workflows."
|
||||
version = "1.1.0"
|
||||
license = { file = "LICENSE" }
|
||||
# Keep in sync with requirements.txt. GitHub/CUDA-flavored extras (vggt,
|
||||
# gsplat) and the ComfyUI-Flux-Inpainting sibling pack are installed by
|
||||
# install.py, which ComfyUI-Manager runs automatically after install.
|
||||
dependencies = [
|
||||
"transformers==4.50.0",
|
||||
"diffusers==0.33.1",
|
||||
"open3d==0.19.0",
|
||||
"protobuf",
|
||||
"click",
|
||||
"timm",
|
||||
"plyfile",
|
||||
"pillow-heif",
|
||||
"matplotlib",
|
||||
"imageio",
|
||||
"imageio-ffmpeg",
|
||||
]
|
||||
|
||||
[project.urls]
|
||||
Repository = "https://github.com/Alexankharin/camera-comfyUI"
|
||||
|
||||
[tool.comfy]
|
||||
PublisherId = "alexk"
|
||||
DisplayName = "camera-comfyUI"
|
||||
# Force-include the SHARP submodule: its files are a gitlink in the parent repo
|
||||
# (not git-tracked files), so without this the registry archive would ship
|
||||
# without submodules/ml-sharpt and ImageToSplat/VideoToFusedSplats would be
|
||||
# unavailable until users clone it manually.
|
||||
includes = ["submodules/ml-sharpt/"]
|
||||
@@ -42,7 +42,14 @@ def map_grid(
|
||||
output_horizontal_fov = torch.tensor(output_horizontal_fov, device=grid_torch.device).float()
|
||||
|
||||
# Calculate vertical field of view for input and output projections
|
||||
output_vertical_fov = output_horizontal_fov # Assuming square aspect ratio
|
||||
# For equirectangular, use 2:1 aspect ratio (vertical FOV = horizontal FOV / 2)
|
||||
if output_projection == "EQUIRECTANGULAR":
|
||||
output_vertical_fov = output_horizontal_fov / 2.0
|
||||
else:
|
||||
output_vertical_fov = output_horizontal_fov # Assuming square aspect ratio for other projections
|
||||
|
||||
# Calculate input vertical FOV based on output grid aspect ratio
|
||||
# This allows the input's vertical range to adapt to the output dimensions
|
||||
input_vertical_fov = input_horizontal_fov * (grid_torch.shape[0] / grid_torch.shape[1])
|
||||
|
||||
# Normalize the grid for vertical FOV adjustment
|
||||
@@ -148,7 +155,7 @@ class ReprojectImage:
|
||||
RETURN_TYPES: Tuple[str, str] = ("IMAGE", "MASK")
|
||||
RETURN_NAMES = ("reprojected image", "reprojected mask")
|
||||
FUNCTION: str = "reproject_image"
|
||||
CATEGORY: str = "Camera/reproject"
|
||||
CATEGORY: str = "Camera/Reprojection"
|
||||
|
||||
def reproject_image(
|
||||
self,
|
||||
@@ -222,8 +229,8 @@ class ReprojectImage:
|
||||
)
|
||||
|
||||
grid_y, grid_x = torch.meshgrid(
|
||||
torch.linspace(-1, 1, output_width, device=image_tensor.device),
|
||||
torch.linspace(-1, 1, output_height, device=image_tensor.device),
|
||||
torch.linspace(-1, 1, output_width, device=image_tensor.device),
|
||||
indexing="ij"
|
||||
)
|
||||
grid_init = torch.stack((grid_x, grid_y), dim=-1)
|
||||
@@ -303,7 +310,7 @@ class TransformToMatrix:
|
||||
RETURN_TYPES: Tuple[str] = ("MAT_4X4",)
|
||||
RETURN_NAMES = ("transformation matrix",)
|
||||
FUNCTION: str = "generate_matrix"
|
||||
CATEGORY: str = "Camera/reproject"
|
||||
CATEGORY: str = "Camera/Matrix"
|
||||
|
||||
def generate_matrix(
|
||||
self,
|
||||
@@ -392,7 +399,7 @@ class TransformToMatrixManual:
|
||||
RETURN_TYPES: Tuple[str] = ("MAT_4X4",)
|
||||
RETURN_NAMES = ("transformation matrix",)
|
||||
FUNCTION: str = "generate_matrix"
|
||||
CATEGORY: str = "Camera/reproject"
|
||||
CATEGORY: str = "Camera/Matrix"
|
||||
|
||||
def generate_matrix(
|
||||
self,
|
||||
@@ -444,7 +451,7 @@ class ReprojectDepth:
|
||||
RETURN_TYPES: Tuple[str, str] = ("TENSOR", "MASK")
|
||||
RETURN_NAMES = ("reprojected_depth", "reprojected_mask")
|
||||
FUNCTION: str = "reproject_depth"
|
||||
CATEGORY: str = "Camera/reproject"
|
||||
CATEGORY: str = "Camera/Reprojection"
|
||||
|
||||
def reproject_depth(
|
||||
self,
|
||||
|
||||
@@ -2,9 +2,17 @@ transformers==4.50.0
|
||||
diffusers==0.33.1
|
||||
open3d==0.19.0
|
||||
protobuf
|
||||
# Runtime deps of the bundled SHARP submodule (submodules/ml-sharpt), so
|
||||
# ImageToSplat & co. work out of the box. torch/torchvision/scipy/tqdm come
|
||||
# with ComfyUI itself; gsplat and vggt are GitHub/CUDA-flavored and are
|
||||
# handled by install.py (run automatically by ComfyUI-Manager).
|
||||
click
|
||||
timm
|
||||
plyfile
|
||||
pillow-heif
|
||||
matplotlib
|
||||
imageio
|
||||
imageio-ffmpeg
|
||||
# open3d is optional, see pointcloud_nodes.py
|
||||
# Python version: 3.12.x (used by embedded python)
|
||||
# All versions pinned to match embedded environment
|
||||
# If using a different Python, adjust versions accordingly
|
||||
# For full reproducibility, consider using a virtual environment
|
||||
# and pip freeze > requirements.txt
|
||||
|
||||
@@ -0,0 +1,299 @@
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
import numpy as np
|
||||
import os
|
||||
import sys
|
||||
from typing import Dict, Any, Tuple
|
||||
from tqdm import tqdm # Added tqdm import
|
||||
|
||||
# Import existing pointcloud nodes and projection definitions
|
||||
from .pointcloud_nodes import DepthToPointCloud, TransformPointCloud, ProjectPointCloud, Projection, PointCloudCleaner, interpolate_se3
|
||||
import folder_paths
|
||||
|
||||
# Ensure video_depth_anything is on path
|
||||
_here = os.path.dirname(os.path.abspath(__file__))
|
||||
|
||||
# climb up 3 levels: camera-comfyUI → custom_nodes → ComfyUI
|
||||
COMFYUI_ROOT = os.path.abspath(os.path.join(_here, os.pardir, os.pardir))
|
||||
|
||||
# point at metric_depth inside the Video-Depth-Anything clone at the ComfyUI root
|
||||
video_depth_path = os.path.join(COMFYUI_ROOT, "Video-Depth-Anything", "metric_depth")
|
||||
|
||||
# insert at front so it always wins
|
||||
if video_depth_path not in sys.path:
|
||||
sys.path.insert(0, video_depth_path)
|
||||
NO_VIDEO_DEPTH_ANYTHING= False
|
||||
try:
|
||||
from video_depth_anything.video_depth import VideoDepthAnything
|
||||
print("✅ video_depth_anything module loaded successfully.")
|
||||
except ImportError as e:
|
||||
NO_VIDEO_DEPTH_ANYTHING = True
|
||||
print(
|
||||
f"❌ Could not load video_depth_anything from {video_depth_path!r}: {e}"
|
||||
)
|
||||
|
||||
class VideoCameraMotionSequence:
|
||||
"""
|
||||
Takes a sequence of RGB frames and corresponding depth maps,
|
||||
converts each frame+depth to a pointcloud, interpolates a camera
|
||||
trajectory to match video length, cleans the pointcloud if needed,
|
||||
and outputs reprojected images, masks, and depth maps per frame.
|
||||
"""
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
# Sequence of frames: Tensor [T, H, W, 3]
|
||||
"frames": ("IMAGE", {"shape_hint": [None, None, None, 3]}),
|
||||
# Sequence of depth maps: Tensor [T, H, W] or [T, H, W, 1]
|
||||
"depth_seq": ("TENSOR", {"shape_hint": [None, None, None]}),
|
||||
# Camera trajectory waypoints: Tensor [K, 4, 4]
|
||||
"trajectory": ("TENSOR", {"shape_hint": [None, 4, 4]}),
|
||||
# Optional mask sequence: Tensor [T, H, W] or [T, H, W, 1]
|
||||
"mask_seq": ("MASK", {"shape_hint": [None, None, None], "optional": True}),
|
||||
# Input projection parameters
|
||||
"input_projection": (Projection.PROJECTIONS, {}),
|
||||
"input_horizontal_fov": ("FLOAT", {"default": 90.0}),
|
||||
"depth_scale": ("FLOAT", {"default": 1.0}),
|
||||
"invert_depth": ("BOOLEAN", {"default": False}),
|
||||
# Output projection parameters
|
||||
"output_projection": (Projection.PROJECTIONS, {}),
|
||||
"output_horizontal_fov": ("FLOAT", {"default": 90.0}),
|
||||
"output_width": ("INT", {"default": 512, "min": 1}),
|
||||
"output_height": ("INT", {"default": 512, "min": 1}),
|
||||
"point_size": ("INT", {"default": 1, "min": 1}),
|
||||
# Cleaning parameters
|
||||
"voxel_size": ("FLOAT", {"default": 1.0, "min": 1e-3}),
|
||||
"min_points_per_voxel": ("INT", {"default": 3, "min": 1}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("IMAGE", "MASK", "TENSOR")
|
||||
RETURN_NAMES = ("video_frames", "mask_frames", "depths")
|
||||
FUNCTION = "process_sequence"
|
||||
CATEGORY = "Camera/Video"
|
||||
|
||||
def process_sequence(
|
||||
self,
|
||||
frames: torch.Tensor,
|
||||
depth_seq: torch.Tensor,
|
||||
trajectory: torch.Tensor,
|
||||
input_projection: str,
|
||||
input_horizontal_fov: float,
|
||||
depth_scale: float,
|
||||
invert_depth: bool,
|
||||
output_projection: str,
|
||||
output_horizontal_fov: float,
|
||||
output_width: int,
|
||||
output_height: int,
|
||||
point_size: int,
|
||||
voxel_size: float,
|
||||
min_points_per_voxel: int,
|
||||
mask_seq: torch.Tensor = None,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
# frames: [T, H, W, 3]
|
||||
# depth_seq: [T, H, W] or [T, H, W, 1]
|
||||
T, H, W, _ = frames.shape
|
||||
|
||||
# Interpolate trajectory to match T (SE(3): quaternion SLERP on R, lerp on t)
|
||||
interp_traj = interpolate_se3(trajectory, T)
|
||||
|
||||
out_frames = []
|
||||
out_masks = []
|
||||
out_depths = []
|
||||
|
||||
# Add tqdm progress bar for the sequence
|
||||
# If mask_seq is a single mask [H, W] or [H, W, 1], repeat it for all frames
|
||||
if mask_seq is not None:
|
||||
if mask_seq.dim() == 2 or (mask_seq.dim() == 3 and mask_seq.shape[0] == 1):
|
||||
mask_seq = mask_seq.unsqueeze(0) if mask_seq.dim() == 2 else mask_seq
|
||||
mask_seq = mask_seq.repeat(T, 1, 1, 1) if mask_seq.dim() == 4 else mask_seq.repeat(T, 1, 1)
|
||||
|
||||
for i, (frame, depth, pose) in enumerate(tqdm(zip(frames, depth_seq, interp_traj), total=T, desc="Processing video frames")):
|
||||
if depth.dim() == 3 and depth.shape[-1] == 1:
|
||||
depth = depth.squeeze(-1)
|
||||
# Use mask if provided; must be (re)initialized every iteration
|
||||
mask = None
|
||||
if mask_seq is not None:
|
||||
mask = mask_seq[i]
|
||||
if mask.dim() == 3 and mask.shape[-1] == 1:
|
||||
mask = mask.squeeze(-1)
|
||||
# to pointcloud
|
||||
pc, = DepthToPointCloud().depth_to_pointcloud(
|
||||
image=frame.permute(2, 0, 1),
|
||||
input_projection=input_projection,
|
||||
input_horizontal_fov=input_horizontal_fov,
|
||||
depth_scale=depth_scale,
|
||||
invert_depth=invert_depth,
|
||||
depthmap=depth,
|
||||
mask=mask,
|
||||
)
|
||||
# optional cleaning
|
||||
if min_points_per_voxel > 1:
|
||||
pc, = PointCloudCleaner().clean_pointcloud(
|
||||
pointcloud=pc,
|
||||
width=output_width,
|
||||
height=output_height,
|
||||
voxel_size=voxel_size,
|
||||
min_points_per_voxel=min_points_per_voxel,
|
||||
)
|
||||
# transform and project
|
||||
pc_t, = TransformPointCloud().transform_pointcloud(pc, pose)
|
||||
img_t, mask_t, depth_t = ProjectPointCloud().project_pointcloud(
|
||||
pointcloud=pc_t,
|
||||
output_projection=output_projection,
|
||||
output_horizontal_fov=output_horizontal_fov,
|
||||
output_width=output_width,
|
||||
output_height=output_height,
|
||||
point_size=point_size,
|
||||
)
|
||||
|
||||
out_frames.append(img_t[0])
|
||||
out_masks.append(mask_t)
|
||||
out_depths.append(depth_t)
|
||||
|
||||
return (
|
||||
torch.stack(out_frames, dim=0), # [T, 3, H, W]
|
||||
torch.stack(out_masks, dim=0), # [T, H, W]
|
||||
torch.stack(out_depths, dim=0), # [T, H, W]
|
||||
)
|
||||
|
||||
|
||||
class DepthFramesToVideo:
|
||||
"""
|
||||
Converts a sequence of depth maps into video frame tensors for saving.
|
||||
"""
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
"depth_seq": ("TENSOR", {"shape_hint": [None, None, None]}),
|
||||
"mask_seq": ("MASK", {"shape_hint": [None, None, None]}),
|
||||
"normalize": ("BOOLEAN", {"default": True}),
|
||||
"invert_depth": ("BOOLEAN", {"default": False}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("TENSOR", "IMAGE")
|
||||
RETURN_NAMES = ("video_frames", "depth_video")
|
||||
FUNCTION = "depth_to_video_frames"
|
||||
CATEGORY = "Camera/Video"
|
||||
|
||||
def depth_to_video_frames(
|
||||
self,
|
||||
depth_seq: torch.Tensor,
|
||||
normalize: bool,
|
||||
invert_depth: bool,
|
||||
mask_seq: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
ds = depth_seq.clone().squeeze()
|
||||
if ds.dim() == 2:
|
||||
ds = ds.unsqueeze(0) # [H, W] -> [1, H, W]
|
||||
if ds.dim() != 3:
|
||||
raise ValueError(f"Expected ds to be 3D [T, H, W], got shape {ds.shape}")
|
||||
if invert_depth:
|
||||
ds= 1.0 / (ds + 1e-8) # Avoid division by zero
|
||||
if normalize:
|
||||
# Mask: only normalize where depth > 0
|
||||
mask = mask_seq>0.5
|
||||
if mask.any():
|
||||
#percentile first 10 percent min
|
||||
# sample
|
||||
minv = ds[mask]
|
||||
# sample 10000 and find 10% quantile
|
||||
if minv.numel() > 10000:
|
||||
minv = minv[torch.randperm(minv.numel())[:10000]]
|
||||
|
||||
minv = minv.quantile(0.2)
|
||||
minv = minv if minv > 0.1 else 0.1 # Avoid division by zero
|
||||
#percentile last 10 percent max
|
||||
maxv = ds[mask]
|
||||
if maxv.numel() > 10000:
|
||||
maxv = maxv[torch.randperm(maxv.numel())[:10000]]
|
||||
maxv = maxv.quantile(0.98)
|
||||
maxv = maxv if maxv < 100 else 100
|
||||
print(f"Normalizing depth: min={minv}, max={maxv}")
|
||||
ds_norm = (ds - minv) / (maxv - minv + 1e-8)
|
||||
ds = ds_norm.clamp(0, 1) # torch.where(mask, ds_norm, ds) # Only normalize valid values
|
||||
else:
|
||||
print("Warning: No valid depth values for normalization.")
|
||||
# expand to 3 channels: [T, H, W] -> [T, 3, H, W]
|
||||
raw = depth_seq.clone().squeeze()
|
||||
ds_u8 = (ds * 255.0).round().to(torch.uint8)
|
||||
raw_u8 = (raw.clamp(0, 255)).to(torch.uint8) # if raw is already in a displayable range
|
||||
|
||||
# expand to 3 channels and permute to HWC
|
||||
ds_color = ds_u8.unsqueeze(1).repeat(1, 3, 1, 1).permute(0, 2, 3, 1)
|
||||
raw_color = raw_u8.unsqueeze(1).repeat(1, 3, 1, 1).permute(0, 2, 3, 1)
|
||||
return raw_color, ds_color # [T, 3, H, W] -> [T, H, W, 3]
|
||||
|
||||
# Cache for loaded VideoDepthAnything models, keyed by (checkpoint, device)
|
||||
_VIDEO_DEPTH_MODEL_CACHE: Dict[Tuple[str, str], Any] = {}
|
||||
|
||||
|
||||
class VideoMetricDepthEstimate:
|
||||
"""
|
||||
Estimates metric depth for a sequence of frames using VideoDepthAnything.
|
||||
"""
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
# model files (.pth) in input directory
|
||||
model_dir = os.path.join(os.getcwd(), "models", "checkpoints")
|
||||
os.makedirs(model_dir, exist_ok=True)
|
||||
files = [f for f in os.listdir(model_dir) if f.lower().endswith(('.pth', '.ckpt', '.safetensors'))]
|
||||
return {
|
||||
"required": {
|
||||
"frames": ("IMAGE", {"shape_hint": [None, None, None, 3]}),
|
||||
"model_checkpoint": (files, {"file_chooser": True}),
|
||||
"input_size": ("INT", {"default": 518, "min": 64, "max": 2048}),
|
||||
"max_fps": ("INT", {"default": 60, "min": 1}),
|
||||
}
|
||||
}
|
||||
RETURN_TYPES = ("TENSOR", "FLOAT")
|
||||
RETURN_NAMES = ("metric_depths", "fps")
|
||||
FUNCTION = "estimate_metric_depth"
|
||||
CATEGORY = "Camera/Video"
|
||||
|
||||
def estimate_metric_depth(
|
||||
self,
|
||||
frames: torch.Tensor,
|
||||
model_checkpoint: str,
|
||||
input_size: int,
|
||||
max_fps: int,
|
||||
) -> Tuple[torch.Tensor, float]:
|
||||
if NO_VIDEO_DEPTH_ANYTHING:
|
||||
raise ImportError(
|
||||
f"VideoDepthAnything library not found. Clone "
|
||||
f"https://github.com/DepthAnything/Video-Depth-Anything into {COMFYUI_ROOT!r} "
|
||||
f"(expected module path: {video_depth_path!r})."
|
||||
)
|
||||
device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
|
||||
# if max input<1.5 normalize to 0-255
|
||||
if frames.max() < 1.5:
|
||||
frames = (frames * 255)
|
||||
cache_key = (model_checkpoint, str(device))
|
||||
model = _VIDEO_DEPTH_MODEL_CACHE.get(cache_key)
|
||||
if model is None:
|
||||
# Same checkpoint directory as computed in INPUT_TYPES
|
||||
model_dir = os.path.join(os.getcwd(), "models", "checkpoints")
|
||||
checkpoint_path = os.path.join(model_dir, model_checkpoint)
|
||||
if not os.path.isfile(checkpoint_path):
|
||||
raise FileNotFoundError(f"Checkpoint not found: {checkpoint_path}")
|
||||
model = VideoDepthAnything(**{"encoder": "vitl", "features": 256, "out_channels": [256,512,1024,1024]})
|
||||
state = torch.load(checkpoint_path, map_location='cpu')
|
||||
model.load_state_dict(state, strict=True)
|
||||
model = model.to(device).eval()
|
||||
_VIDEO_DEPTH_MODEL_CACHE[cache_key] = model
|
||||
np_frames = frames.cpu().numpy().astype(np.uint8)
|
||||
metric_depths, fps = model.infer_video_depth(np_frames, max_fps, input_size=input_size, device=device.type, fp32=False)
|
||||
return (torch.from_numpy(metric_depths), float(fps))
|
||||
|
||||
# Register nodes
|
||||
if NO_VIDEO_DEPTH_ANYTHING:
|
||||
NODE_CLASS_MAPPINGS = {}
|
||||
else:
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"VideoCameraMotionSequence": VideoCameraMotionSequence,
|
||||
"VideoMetricDepthEstimate": VideoMetricDepthEstimate,
|
||||
"DepthFramesToVideo": DepthFramesToVideo,
|
||||
}
|
||||
|
After Width: | Height: | Size: 40 KiB |
|
After Width: | Height: | Size: 27 KiB |
@@ -1 +1,289 @@
|
||||
{"id":"8c6e5ec1-4ff2-42d9-9408-fcd0a23d362a","revision":0,"last_node_id":11,"last_link_id":17,"nodes":[{"id":3,"type":"PreviewImage","pos":[-5.768195629119873,204.72561645507812],"size":[210,246],"flags":{},"order":2,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":16}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":4,"type":"MaskToImage","pos":[-30.901016235351562,532.864013671875],"size":[264.5999755859375,26],"flags":{},"order":3,"mode":0,"inputs":[{"localized_name":"mask","name":"mask","type":"MASK","link":17}],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[4]}],"properties":{"Node name for S&R":"MaskToImage"},"widgets_values":[]},{"id":2,"type":"LoadImage","pos":[-843.689208984375,354.9311828613281],"size":[315,314],"flags":{},"order":0,"mode":0,"inputs":[],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[15]},{"localized_name":"MASK","name":"MASK","type":"MASK","links":[]}],"properties":{"Node name for S&R":"LoadImage"},"widgets_values":["flux_dev_example.png","image",""]},{"id":5,"type":"PreviewImage","pos":[296.6977844238281,611.6702880859375],"size":[210,246],"flags":{},"order":4,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":4}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":11,"type":"OutpaintAnyProjection","pos":[-498.9301452636719,317.144287109375],"size":[400,492],"flags":{},"order":1,"mode":0,"inputs":[{"localized_name":"image","name":"image","type":"IMAGE","link":15},{"localized_name":"mask","name":"mask","shape":7,"type":"MASK","link":null}],"outputs":[{"localized_name":"final_image","name":"final_image","type":"IMAGE","links":[16]},{"localized_name":"needs_inpaint_mask","name":"needs_inpaint_mask","type":"MASK","links":[17]}],"properties":{"Node name for S&R":"OutpaintAnyProjection"},"widgets_values":["PINHOLE",90,"FISHEYE",180,4096,4096,"PINHOLE",90,1024,45,0,"",10,false,30,1,false]}],"links":[[4,4,0,5,0,"IMAGE"],[15,2,0,11,0,"IMAGE"],[16,11,0,3,0,"IMAGE"],[17,11,1,4,0,"MASK"]],"groups":[],"config":{},"extra":{"ds":{"scale":0.8390545288824094,"offset":[703.1897327574803,-308.43624510132975]}},"version":0.4}
|
||||
{
|
||||
"id": "8c6e5ec1-4ff2-42d9-9408-fcd0a23d362a",
|
||||
"revision": 0,
|
||||
"last_node_id": 12,
|
||||
"last_link_id": 17,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 3,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
-5.768195629119873,
|
||||
204.72561645507812
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 16
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
],
|
||||
"title": "Result"
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "MaskToImage",
|
||||
"pos": [
|
||||
-30.901016235351562,
|
||||
532.864013671875
|
||||
],
|
||||
"size": [
|
||||
264.5999755859375,
|
||||
26
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"type": "MASK",
|
||||
"link": 17
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
4
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "MaskToImage"
|
||||
},
|
||||
"widgets_values": []
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "LoadImage",
|
||||
"pos": [
|
||||
-843.689208984375,
|
||||
354.9311828613281
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
15
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": []
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"camera_example_pinhole.jpg",
|
||||
"image",
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
296.6977844238281,
|
||||
611.6702880859375
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 4
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
],
|
||||
"title": "Holes still to fill"
|
||||
},
|
||||
{
|
||||
"id": 11,
|
||||
"type": "OutpaintAnyProjection",
|
||||
"pos": [
|
||||
-498.9301452636719,
|
||||
317.144287109375
|
||||
],
|
||||
"size": [
|
||||
400,
|
||||
492
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 15
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"shape": 7,
|
||||
"type": "MASK",
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "final_image",
|
||||
"name": "final_image",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
16
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "needs_inpaint_mask",
|
||||
"name": "needs_inpaint_mask",
|
||||
"type": "MASK",
|
||||
"links": [
|
||||
17
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "OutpaintAnyProjection"
|
||||
},
|
||||
"widgets_values": [
|
||||
"PINHOLE",
|
||||
90,
|
||||
"FISHEYE",
|
||||
180,
|
||||
4096,
|
||||
4096,
|
||||
"PINHOLE",
|
||||
90,
|
||||
1024,
|
||||
45,
|
||||
0,
|
||||
"",
|
||||
10,
|
||||
false,
|
||||
30,
|
||||
1,
|
||||
false
|
||||
],
|
||||
"title": "Outpaint one patch (yaw +45°)"
|
||||
},
|
||||
{
|
||||
"id": 12,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
-1403.689208984375,
|
||||
204.72561645507812
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
288
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# OutpaintAnyProjection smoke test\n\nSingle-node sanity check: a 90° pinhole image is placed on a 180° fisheye canvas and one 90° pinhole patch at yaw +45° is Flux-inpainted (10 steps for speed).\n\nOutputs: the partially outpainted canvas and the *remaining holes* mask — chain more OutpaintAnyProjection nodes at other angles to fill it (see `Outpaint_fisheye180.json`).\n\n- **Set:** input image; prompt inside the node (empty = unconditional).\n- **Requires:** `custom_nodes/inpainting_flux` (installed automatically by this pack's install.py) — downloads FLUX.1-Fill NF4 weights on first run."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
4,
|
||||
4,
|
||||
0,
|
||||
5,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
15,
|
||||
2,
|
||||
0,
|
||||
11,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
16,
|
||||
11,
|
||||
0,
|
||||
3,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
17,
|
||||
11,
|
||||
1,
|
||||
4,
|
||||
0,
|
||||
"MASK"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.8390545288824094,
|
||||
"offset": [
|
||||
703.1897327574803,
|
||||
-308.43624510132975
|
||||
]
|
||||
},
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"id": "312a3a27-6189-4cd7-8c9f-5a62881c623e",
|
||||
"revision": 0,
|
||||
"last_node_id": 18,
|
||||
"last_node_id": 19,
|
||||
"last_link_id": 37,
|
||||
"nodes": [
|
||||
{
|
||||
@@ -38,7 +38,7 @@
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Fisheye_outpainted_flux_dev.png",
|
||||
"camera_example_fisheye.jpg",
|
||||
"image"
|
||||
]
|
||||
},
|
||||
@@ -213,7 +213,9 @@
|
||||
90,
|
||||
1024,
|
||||
4096,
|
||||
21
|
||||
"SOFTMERGE",
|
||||
21,
|
||||
1
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -326,14 +328,11 @@
|
||||
10,
|
||||
7.5,
|
||||
5,
|
||||
"PINHOLE",
|
||||
90,
|
||||
1024,
|
||||
1024,
|
||||
"open",
|
||||
9,
|
||||
15
|
||||
]
|
||||
0.07,
|
||||
3,
|
||||
"Depth-Anything-V2-Metric-Indoor-Base-hf"
|
||||
],
|
||||
"title": "Enrich cloud along trajectory (Flux outpaint)"
|
||||
},
|
||||
{
|
||||
"id": 18,
|
||||
@@ -411,8 +410,36 @@
|
||||
90,
|
||||
1024,
|
||||
1024,
|
||||
2
|
||||
]
|
||||
2,
|
||||
0,
|
||||
false,
|
||||
false
|
||||
],
|
||||
"title": "Orbit preview of enriched cloud"
|
||||
},
|
||||
{
|
||||
"id": 19,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
982.239013671875,
|
||||
1182.49755859375
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Point cloud enricher (outpaint along a trajectory)\n\nFisheye image → metric depth → point cloud, then **PointcloudTrajectoryEnricher** walks the loaded camera trajectory: at each pose it renders the cloud, Flux-inpaints the disocclusion holes, re-estimates depth, aligns it and merges the new points into the cloud. The enriched cloud is saved and previewed as an orbit video.\n\nThis is the one-node version of `pointcloud_inpaint.json`.\n\n- **Set:** input image; trajectory .npy (record one with SaveTrajectory); prompt inside the enricher.\n- **Requires:** inpainting_flux (auto-installed), Depth-Anything V2, FLUX.1-Fill NF4 weights.\n- **Note:** 2026-07 schema migration — the enricher's render/reproject options are now internal; voxel merge params were reset to defaults."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
@@ -513,7 +540,47 @@
|
||||
"TENSOR"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"title": "1. Fisheye → point cloud",
|
||||
"bounding": [
|
||||
1522.239013671875,
|
||||
1122.49755859375,
|
||||
1332.4192962646484,
|
||||
589.1494140625
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"title": "2. Enrich along trajectory",
|
||||
"bounding": [
|
||||
1964.2667236328125,
|
||||
1479.796630859375,
|
||||
1073.373291015625,
|
||||
614.0
|
||||
],
|
||||
"color": "#a1309b",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"title": "3. Save & preview",
|
||||
"bounding": [
|
||||
2283.51123046875,
|
||||
1545.8380126953125,
|
||||
1130.08935546875,
|
||||
726.26318359375
|
||||
],
|
||||
"color": "#8A8",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
}
|
||||
],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
@@ -523,7 +590,8 @@
|
||||
-1433.3959538925521
|
||||
]
|
||||
},
|
||||
"frontendVersion": "1.19.9"
|
||||
"frontendVersion": "1.19.9",
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
}
|
||||
|
||||
|
After Width: | Height: | Size: 65 KiB |
@@ -1 +0,0 @@
|
||||
{"id":"dd56c0bf-7405-406e-924f-42b2feacb73f","revision":0,"last_node_id":6,"last_link_id":4,"nodes":[{"id":1,"type":"TransformToMatrix","pos":[-337.9580993652344,1500.5211181640625],"size":[315,154],"flags":{},"order":0,"mode":0,"inputs":[],"outputs":[{"localized_name":"MAT_4X4","name":"MAT_4X4","type":"MAT_4X4","links":[1]}],"properties":{"Node name for S&R":"TransformToMatrix"},"widgets_values":[0,0,0,0,0]},{"id":2,"type":"CameraMotion","pos":[130.0218505859375,1516.7567138671875],"size":[367.79998779296875,218],"flags":{},"order":3,"mode":0,"inputs":[{"localized_name":"pointcloud","name":"pointcloud","type":"TENSOR","link":4},{"localized_name":"initial_matrix","name":"initial_matrix","type":"MAT_4X4","link":1},{"localized_name":"final_matrix","name":"final_matrix","type":"MAT_4X4","link":2}],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[3]}],"properties":{"Node name for S&R":"CameraMotion"},"widgets_values":[24,"PINHOLE",90,1024,1024,2]},{"id":3,"type":"SaveWEBM","pos":[606.5880737304688,1515.5966796875],"size":[315,437],"flags":{},"order":4,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":3}],"outputs":[],"properties":{},"widgets_values":["ComfyUI","vp9",10.000000000000002,32]},{"id":5,"type":"TransformToMatrix","pos":[-262.9134521484375,1740.8876953125],"size":[315,154],"flags":{},"order":1,"mode":0,"inputs":[],"outputs":[{"localized_name":"MAT_4X4","name":"MAT_4X4","type":"MAT_4X4","links":[2]}],"properties":{"Node name for S&R":"TransformToMatrix"},"widgets_values":[0.10000000000000002,0,0,0,0]},{"id":6,"type":"LoadPointCloud","pos":[-357.24761962890625,1294.427734375],"size":[315,58],"flags":{},"order":2,"mode":0,"inputs":[],"outputs":[{"localized_name":"TENSOR","name":"TENSOR","type":"TENSOR","links":[4]}],"properties":{"Node name for S&R":"LoadPointCloud"},"widgets_values":["ComfyUIPointCloud_00001.ply"]}],"links":[[1,1,0,2,1,"MAT_4X4"],[2,5,0,2,2,"MAT_4X4"],[3,2,0,3,0,"IMAGE"],[4,6,0,2,0,"TENSOR"]],"groups":[],"config":{},"extra":{"ds":{"scale":1.351305709310409,"offset":[-187.6257577580669,-1567.2143321744395]}},"version":0.4}
|
||||
@@ -1,2 +0,0 @@
|
||||
# This file marks the workflows directory as a Python package.
|
||||
NODE_CLASS_MAPPINGS={}
|
||||
|
After Width: | Height: | Size: 38 KiB |
@@ -1 +1,331 @@
|
||||
{"id":"de69fe58-2f01-4527-92b1-681a151b1ce3","revision":0,"last_node_id":7,"last_link_id":8,"nodes":[{"id":3,"type":"LoadImage","pos":[-897.7078247070312,314.2335205078125],"size":[315,314],"flags":{},"order":0,"mode":0,"inputs":[],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[2]},{"localized_name":"MASK","name":"MASK","type":"MASK","links":null}],"properties":{"Node name for S&R":"LoadImage"},"widgets_values":["example.png","image",""]},{"id":1,"type":"TransformToMatrix","pos":[-871.939697265625,694.5091552734375],"size":[315,154],"flags":{},"order":1,"mode":0,"inputs":[],"outputs":[{"localized_name":"MAT_4X4","name":"MAT_4X4","type":"MAT_4X4","links":[5]}],"properties":{"Node name for S&R":"TransformToMatrix"},"widgets_values":[0,0,0,60,0]},{"id":6,"type":"MaskToImage","pos":[-169.98599243164062,763.4010009765625],"size":[264.5999755859375,26],"flags":{},"order":4,"mode":0,"inputs":[{"localized_name":"mask","name":"mask","type":"MASK","link":6}],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[7]}],"properties":{"Node name for S&R":"MaskToImage"}},{"id":4,"type":"PreviewImage","pos":[299.262451171875,609.4740600585938],"size":[210,246],"flags":{},"order":5,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":7}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":7,"type":"PreviewImage","pos":[-14.518107414245605,366.5712585449219],"size":[210,246],"flags":{},"order":3,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":8}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":2,"type":"ReprojectImage","pos":[-468.4571838378906,358.2191467285156],"size":[315,266],"flags":{},"order":2,"mode":0,"inputs":[{"localized_name":"image","name":"image","type":"IMAGE","link":2},{"localized_name":"mask","name":"mask","shape":7,"type":"MASK","link":null},{"localized_name":"transform_matrix","name":"transform_matrix","shape":7,"type":"MAT_4X4","link":5}],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[8]},{"localized_name":"MASK","name":"MASK","type":"MASK","links":[6]}],"properties":{"Node name for S&R":"ReprojectImage"},"widgets_values":[90,180,"PINHOLE","EQUIRECTANGULAR",1024,1024,false,7]}],"links":[[2,3,0,2,0,"IMAGE"],[5,1,0,2,2,"MAT_4X4"],[6,2,1,6,0,"MASK"],[7,6,0,4,0,"IMAGE"],[8,2,0,7,0,"IMAGE"]],"groups":[],"config":{},"extra":{"ds":{"scale":0.8264462809917362,"offset":[806.6857406850668,-309.49182798364893]}},"version":0.4}
|
||||
{
|
||||
"id": "de69fe58-2f01-4527-92b1-681a151b1ce3",
|
||||
"revision": 0,
|
||||
"last_node_id": 8,
|
||||
"last_link_id": 8,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 3,
|
||||
"type": "LoadImage",
|
||||
"pos": [
|
||||
-897.7078247070312,
|
||||
314.2335205078125
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
2
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"camera_example_pinhole.jpg",
|
||||
"image",
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 1,
|
||||
"type": "TransformToMatrix",
|
||||
"pos": [
|
||||
-871.939697265625,
|
||||
694.5091552734375
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "MAT_4X4",
|
||||
"name": "MAT_4X4",
|
||||
"type": "MAT_4X4",
|
||||
"links": [
|
||||
5
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformToMatrix"
|
||||
},
|
||||
"widgets_values": [
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
60,
|
||||
0
|
||||
],
|
||||
"title": "Rotate camera (pitch 60°)"
|
||||
},
|
||||
{
|
||||
"id": 6,
|
||||
"type": "MaskToImage",
|
||||
"pos": [
|
||||
-169.98599243164062,
|
||||
763.4010009765625
|
||||
],
|
||||
"size": [
|
||||
264.5999755859375,
|
||||
26
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"type": "MASK",
|
||||
"link": 6
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
7
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "MaskToImage"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
299.262451171875,
|
||||
609.4740600585938
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 7
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
],
|
||||
"title": "Coverage mask (white = hole)"
|
||||
},
|
||||
{
|
||||
"id": 7,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
-14.518107414245605,
|
||||
366.5712585449219
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 8
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
],
|
||||
"title": "Reprojected image"
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "ReprojectImage",
|
||||
"pos": [
|
||||
-468.4571838378906,
|
||||
358.2191467285156
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
266
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 2
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"shape": 7,
|
||||
"type": "MASK",
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "transform_matrix",
|
||||
"name": "transform_matrix",
|
||||
"shape": 7,
|
||||
"type": "MAT_4X4",
|
||||
"link": 5
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
8
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": [
|
||||
6
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "ReprojectImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
90,
|
||||
180,
|
||||
"PINHOLE",
|
||||
"EQUIRECTANGULAR",
|
||||
1024,
|
||||
1024,
|
||||
false,
|
||||
7
|
||||
],
|
||||
"title": "Pinhole 90° → Equirect 180°"
|
||||
},
|
||||
{
|
||||
"id": 8,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
-1457.7078247070312,
|
||||
314.2335205078125
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Camera reprojection demo\n\nMinimal example of the two core nodes: **TransformToMatrix** rotates the virtual camera (theta = 60° pitch) and **ReprojectImage** converts a 90° pinhole image into a 180° equirectangular view.\n\nPreviews show the reprojected image and the coverage mask (white = pixels the source image cannot see).\n\n- **Set:** the image in LoadImage.\n- **Requires:** nothing beyond this pack (no models).\n- **Try:** switch `output_projection` to FISHEYE, or raise `feathering` to soften the mask edge."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
2,
|
||||
3,
|
||||
0,
|
||||
2,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
5,
|
||||
1,
|
||||
0,
|
||||
2,
|
||||
2,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
6,
|
||||
2,
|
||||
1,
|
||||
6,
|
||||
0,
|
||||
"MASK"
|
||||
],
|
||||
[
|
||||
7,
|
||||
6,
|
||||
0,
|
||||
4,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
8,
|
||||
2,
|
||||
0,
|
||||
7,
|
||||
0,
|
||||
"IMAGE"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.8264462809917362,
|
||||
"offset": [
|
||||
806.6857406850668,
|
||||
-309.49182798364893
|
||||
]
|
||||
},
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
|
||||
|
After Width: | Height: | Size: 16 KiB |
@@ -1 +1,469 @@
|
||||
{"id":"8e48e6c6-735b-4131-9b4d-5847d2948c18","revision":0,"last_node_id":161,"last_link_id":330,"nodes":[{"id":156,"type":"DepthToImageNode","pos":[101.64265441894531,607.9051513671875],"size":[315,58],"flags":{},"order":2,"mode":0,"inputs":[{"localized_name":"depth","name":"depth","type":"TENSOR","link":323}],"outputs":[{"localized_name":"depth image","name":"depth image","type":"IMAGE","links":[322]}],"properties":{"Node name for S&R":"DepthToImageNode"},"widgets_values":[false]},{"id":157,"type":"PreviewImage","pos":[449.5372314453125,592.8480224609375],"size":[210,246],"flags":{},"order":5,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":322}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":159,"type":"PreviewImage","pos":[128.622802734375,835.9542236328125],"size":[210,246],"flags":{},"order":6,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":326}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":158,"type":"MaskToImage","pos":[105.04651641845703,715.2989501953125],"size":[264.5999755859375,26],"flags":{},"order":3,"mode":0,"inputs":[{"localized_name":"mask","name":"mask","type":"MASK","link":325}],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[326]}],"properties":{"Node name for S&R":"MaskToImage"}},{"id":154,"type":"LoadImage","pos":[-601.5926513671875,610.8336181640625],"size":[315,314],"flags":{},"order":0,"mode":0,"inputs":[],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[324,327]},{"localized_name":"MASK","name":"MASK","type":"MASK","links":null}],"properties":{"Node name for S&R":"LoadImage"},"widgets_values":["Saved_fisheye_00001_.png","image",""]},{"id":155,"type":"FisheyeDepthEstimator","pos":[-258.0684814453125,592.8478393554688],"size":[315,198],"flags":{},"order":1,"mode":0,"inputs":[{"localized_name":"image","name":"image","type":"IMAGE","link":324}],"outputs":[{"localized_name":"depthmap","name":"depthmap","type":"TENSOR","links":[323,329]},{"localized_name":"mask","name":"mask","type":"MASK","links":[325,328]}],"properties":{"Node name for S&R":"FisheyeDepthEstimator"},"widgets_values":["Depth-Anything-V2-Metric-Indoor-Base-hf",1,90,1024,4096,25]},{"id":160,"type":"DepthToPointCloud","pos":[398.363525390625,895.5885620117188],"size":[315,170],"flags":{},"order":4,"mode":0,"inputs":[{"localized_name":"image","name":"image","type":"IMAGE","link":327},{"localized_name":"depthmap","name":"depthmap","shape":7,"type":"TENSOR","link":329},{"localized_name":"mask","name":"mask","shape":7,"type":"MASK","link":328}],"outputs":[{"localized_name":"pointcloud","name":"pointcloud","type":"TENSOR","links":[330]}],"properties":{"Node name for S&R":"DepthToPointCloud"},"widgets_values":["FISHEYE",180,1,false]},{"id":161,"type":"SavePointCloud","pos":[785.9862670898438,881.0266723632812],"size":[315,82],"flags":{},"order":7,"mode":0,"inputs":[{"localized_name":"pointcloud","name":"pointcloud","type":"TENSOR","link":330}],"outputs":[],"properties":{"Node name for S&R":"SavePointCloud"},"widgets_values":["kitchen","npy"]}],"links":[[322,156,0,157,0,"IMAGE"],[323,155,0,156,0,"TENSOR"],[324,154,0,155,0,"IMAGE"],[325,155,1,158,0,"MASK"],[326,158,0,159,0,"IMAGE"],[327,154,0,160,0,"IMAGE"],[328,155,1,160,2,"MASK"],[329,155,0,160,1,"TENSOR"],[330,160,0,161,0,"TENSOR"]],"groups":[],"config":{},"extra":{"ds":{"scale":0.9229599817706441,"offset":[115.11465258583665,-662.5990021779805]},"frontendVersion":"1.18.9"},"version":0.4}
|
||||
{
|
||||
"id": "8e48e6c6-735b-4131-9b4d-5847d2948c18",
|
||||
"revision": 0,
|
||||
"last_node_id": 162,
|
||||
"last_link_id": 330,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 156,
|
||||
"type": "DepthToImageNode",
|
||||
"pos": [
|
||||
101.64265441894531,
|
||||
607.9051513671875
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
58
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "depth",
|
||||
"name": "depth",
|
||||
"type": "TENSOR",
|
||||
"link": 323
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "depth image",
|
||||
"name": "depth image",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
322
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "DepthToImageNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
false
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 157,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
449.5372314453125,
|
||||
592.8480224609375
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 322
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 159,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
128.622802734375,
|
||||
835.9542236328125
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 6,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 326
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 158,
|
||||
"type": "MaskToImage",
|
||||
"pos": [
|
||||
105.04651641845703,
|
||||
715.2989501953125
|
||||
],
|
||||
"size": [
|
||||
264.5999755859375,
|
||||
26
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"type": "MASK",
|
||||
"link": 325
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
326
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "MaskToImage"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 154,
|
||||
"type": "LoadImage",
|
||||
"pos": [
|
||||
-601.5926513671875,
|
||||
610.8336181640625
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
324,
|
||||
327
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"camera_example_fisheye.jpg",
|
||||
"image",
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 155,
|
||||
"type": "FisheyeDepthEstimator",
|
||||
"pos": [
|
||||
-258.0684814453125,
|
||||
592.8478393554688
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
198
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 324
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "depthmap",
|
||||
"name": "depthmap",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
323,
|
||||
329
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"type": "MASK",
|
||||
"links": [
|
||||
325,
|
||||
328
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "FisheyeDepthEstimator"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Depth-Anything-V2-Metric-Indoor-Base-hf",
|
||||
1,
|
||||
90,
|
||||
1024,
|
||||
4096,
|
||||
"SOFTMERGE",
|
||||
25,
|
||||
1
|
||||
],
|
||||
"title": "Metric depth (fisheye-aware)"
|
||||
},
|
||||
{
|
||||
"id": 160,
|
||||
"type": "DepthToPointCloud",
|
||||
"pos": [
|
||||
398.363525390625,
|
||||
895.5885620117188
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
170
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 327
|
||||
},
|
||||
{
|
||||
"localized_name": "depthmap",
|
||||
"name": "depthmap",
|
||||
"shape": 7,
|
||||
"type": "TENSOR",
|
||||
"link": 329
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"shape": 7,
|
||||
"type": "MASK",
|
||||
"link": 328
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
330
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "DepthToPointCloud"
|
||||
},
|
||||
"widgets_values": [
|
||||
"FISHEYE",
|
||||
180,
|
||||
1,
|
||||
false
|
||||
],
|
||||
"title": "Depth → point cloud"
|
||||
},
|
||||
{
|
||||
"id": 161,
|
||||
"type": "SavePointCloud",
|
||||
"pos": [
|
||||
785.9862670898438,
|
||||
881.0266723632812
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
82
|
||||
],
|
||||
"flags": {},
|
||||
"order": 7,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"link": 330
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "SavePointCloud"
|
||||
},
|
||||
"widgets_values": [
|
||||
"kitchen",
|
||||
"npy"
|
||||
],
|
||||
"title": "Save .ply / .npy"
|
||||
},
|
||||
{
|
||||
"id": 162,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
-1161.5926513671875,
|
||||
592.8478393554688
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Fisheye 180° → point cloud\n\nEstimates metric depth directly on a 180° fisheye image — **FisheyeDepthEstimator** internally splits it into pinhole views, runs Depth-Anything V2 on each and merges the depths back (SOFTMERGE) — then unprojects image + depth to a 3D point cloud and saves it.\n\nPreviews: colorized depth and the validity mask.\n\n- **Set:** fisheye input image (e.g. produced by `Outpaint_fisheye180.json`); filename in SavePointCloud.\n- **Requires:** Depth-Anything V2 (auto-downloads from HuggingFace).\n- **Next:** view the cloud with `test_pointcloud_loading.json` or synthesize a second eye with `sbs180_workflow.json`."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
322,
|
||||
156,
|
||||
0,
|
||||
157,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
323,
|
||||
155,
|
||||
0,
|
||||
156,
|
||||
0,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
324,
|
||||
154,
|
||||
0,
|
||||
155,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
325,
|
||||
155,
|
||||
1,
|
||||
158,
|
||||
0,
|
||||
"MASK"
|
||||
],
|
||||
[
|
||||
326,
|
||||
158,
|
||||
0,
|
||||
159,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
327,
|
||||
154,
|
||||
0,
|
||||
160,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
328,
|
||||
155,
|
||||
1,
|
||||
160,
|
||||
2,
|
||||
"MASK"
|
||||
],
|
||||
[
|
||||
329,
|
||||
155,
|
||||
0,
|
||||
160,
|
||||
1,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
330,
|
||||
160,
|
||||
0,
|
||||
161,
|
||||
0,
|
||||
"TENSOR"
|
||||
]
|
||||
],
|
||||
"groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"title": "1. Fisheye metric depth",
|
||||
"bounding": [
|
||||
-621.5926513671875,
|
||||
532.8478393554688,
|
||||
1301.1298828125,
|
||||
569.1063842773438
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"title": "2. Unproject & save",
|
||||
"bounding": [
|
||||
378.363525390625,
|
||||
821.0266723632812,
|
||||
742.6227416992188,
|
||||
264.5618896484375
|
||||
],
|
||||
"color": "#a1309b",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
}
|
||||
],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.9229599817706441,
|
||||
"offset": [
|
||||
115.11465258583665,
|
||||
-662.5990021779805
|
||||
]
|
||||
},
|
||||
"frontendVersion": "1.18.9",
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
|
||||
|
After Width: | Height: | Size: 38 KiB |
|
After Width: | Height: | Size: 50 KiB |
@@ -0,0 +1,781 @@
|
||||
{
|
||||
"id": "016c13e5-c6a7-4b2e-adfc-18e9e969f21e",
|
||||
"revision": 0,
|
||||
"last_node_id": 21,
|
||||
"last_link_id": 32,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 10,
|
||||
"type": "SaveWEBM",
|
||||
"pos": [
|
||||
3352.103515625,
|
||||
1535.6951904296875
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
437
|
||||
],
|
||||
"flags": {},
|
||||
"order": 8,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 24
|
||||
},
|
||||
{
|
||||
"localized_name": "filename_prefix",
|
||||
"name": "filename_prefix",
|
||||
"type": "STRING",
|
||||
"widget": {
|
||||
"name": "filename_prefix"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "codec",
|
||||
"name": "codec",
|
||||
"type": "COMBO",
|
||||
"widget": {
|
||||
"name": "codec"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "fps",
|
||||
"name": "fps",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "fps"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "crf",
|
||||
"name": "crf",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "crf"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"ComfyUI",
|
||||
"vp9",
|
||||
24,
|
||||
32
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "TransformPointCloud",
|
||||
"pos": [
|
||||
2458.775390625,
|
||||
1405.860595703125
|
||||
],
|
||||
"size": [
|
||||
329.20001220703125,
|
||||
46
|
||||
],
|
||||
"flags": {},
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"link": 4
|
||||
},
|
||||
{
|
||||
"localized_name": "transform_matrix",
|
||||
"name": "transform_matrix",
|
||||
"type": "MAT_4X4",
|
||||
"link": 5
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "transformed pointcloud",
|
||||
"name": "transformed pointcloud",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
23
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformPointCloud"
|
||||
},
|
||||
"widgets_values": []
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "TransformToMatrix",
|
||||
"pos": [
|
||||
2296.978759765625,
|
||||
1742.214599609375
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "shiftX",
|
||||
"name": "shiftX",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "shiftX"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "shiftY",
|
||||
"name": "shiftY",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "shiftY"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "shiftZ",
|
||||
"name": "shiftZ",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "shiftZ"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "theta",
|
||||
"name": "theta",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "theta"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "phi",
|
||||
"name": "phi",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "phi"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "transformation matrix",
|
||||
"name": "transformation matrix",
|
||||
"type": "MAT_4X4",
|
||||
"links": [
|
||||
5,
|
||||
25
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformToMatrix"
|
||||
},
|
||||
"widgets_values": [
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0
|
||||
],
|
||||
"title": "Camera move (edit me)"
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "DepthToPointCloud",
|
||||
"pos": [
|
||||
2099.81640625,
|
||||
1385.9761962890625
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
170
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 1
|
||||
},
|
||||
{
|
||||
"localized_name": "depthmap",
|
||||
"name": "depthmap",
|
||||
"shape": 7,
|
||||
"type": "TENSOR",
|
||||
"link": 32
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"shape": 7,
|
||||
"type": "MASK",
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "input_projection",
|
||||
"name": "input_projection",
|
||||
"type": "COMBO",
|
||||
"widget": {
|
||||
"name": "input_projection"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "input_horizontal_fov",
|
||||
"name": "input_horizontal_fov",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "input_horizontal_fov"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "depth_scale",
|
||||
"name": "depth_scale",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "depth_scale"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "invert_depth",
|
||||
"name": "invert_depth",
|
||||
"type": "BOOLEAN",
|
||||
"widget": {
|
||||
"name": "invert_depth"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
4,
|
||||
26
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "DepthToPointCloud"
|
||||
},
|
||||
"widgets_values": [
|
||||
"PINHOLE",
|
||||
60,
|
||||
1,
|
||||
false
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 14,
|
||||
"type": "CameraMotionNode",
|
||||
"pos": [
|
||||
2984.30224609375,
|
||||
1486.6483154296875
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
198
|
||||
],
|
||||
"flags": {},
|
||||
"order": 7,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"link": 23
|
||||
},
|
||||
{
|
||||
"localized_name": "trajectory",
|
||||
"name": "trajectory",
|
||||
"type": "TENSOR",
|
||||
"link": 27
|
||||
},
|
||||
{
|
||||
"localized_name": "n_points",
|
||||
"name": "n_points",
|
||||
"type": "INT",
|
||||
"widget": {
|
||||
"name": "n_points"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "output_projection",
|
||||
"name": "output_projection",
|
||||
"type": "COMBO",
|
||||
"widget": {
|
||||
"name": "output_projection"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "output_horizontal_fov",
|
||||
"name": "output_horizontal_fov",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "output_horizontal_fov"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "output_width",
|
||||
"name": "output_width",
|
||||
"type": "INT",
|
||||
"widget": {
|
||||
"name": "output_width"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "output_height",
|
||||
"name": "output_height",
|
||||
"type": "INT",
|
||||
"widget": {
|
||||
"name": "output_height"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "point_size",
|
||||
"name": "point_size",
|
||||
"type": "INT",
|
||||
"widget": {
|
||||
"name": "point_size"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "motion_frames",
|
||||
"name": "motion_frames",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
24
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "CameraMotionNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
10,
|
||||
"PINHOLE",
|
||||
60,
|
||||
1024,
|
||||
1024,
|
||||
2,
|
||||
0,
|
||||
false,
|
||||
false
|
||||
],
|
||||
"title": "Render fly-through"
|
||||
},
|
||||
{
|
||||
"id": 17,
|
||||
"type": "CameraTrajectoryNode",
|
||||
"pos": [
|
||||
2722.265380859375,
|
||||
1772.097412109375
|
||||
],
|
||||
"size": [
|
||||
317.4000244140625,
|
||||
46
|
||||
],
|
||||
"flags": {},
|
||||
"order": 6,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"link": 26
|
||||
},
|
||||
{
|
||||
"localized_name": "initial_matrix",
|
||||
"name": "initial_matrix",
|
||||
"type": "MAT_4X4",
|
||||
"link": 25
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "trajectory",
|
||||
"name": "trajectory",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
27
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "CameraTrajectoryNode"
|
||||
},
|
||||
"widgets_values": [],
|
||||
"title": "Build trajectory"
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"type": "DepthEstimatorNode",
|
||||
"pos": [
|
||||
1570.1402587890625,
|
||||
1859.879638671875
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
82
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 3
|
||||
},
|
||||
{
|
||||
"localized_name": "model_name",
|
||||
"name": "model_name",
|
||||
"type": "STRING",
|
||||
"widget": {
|
||||
"name": "model_name"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "depth_scale",
|
||||
"name": "depth_scale",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "depth_scale"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "depth tensor",
|
||||
"name": "depth tensor",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
31
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "DepthEstimatorNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Depth-Anything-V2-Metric-Indoor-Base-hf",
|
||||
1.0000000000000002,
|
||||
1
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 20,
|
||||
"type": "ZDepthToRayDepthNode",
|
||||
"pos": [
|
||||
1969.0958251953125,
|
||||
1782.2877197265625
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
58
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "depth",
|
||||
"name": "depth",
|
||||
"type": "TENSOR",
|
||||
"link": 31
|
||||
},
|
||||
{
|
||||
"localized_name": "fov",
|
||||
"name": "fov",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "fov"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "ray depth",
|
||||
"name": "ray depth",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
32
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "ZDepthToRayDepthNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
90
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 1,
|
||||
"type": "LoadImage",
|
||||
"pos": [
|
||||
1575.78955078125,
|
||||
1419.3006591796875
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "COMBO",
|
||||
"widget": {
|
||||
"name": "image"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "choose file to upload",
|
||||
"name": "upload",
|
||||
"type": "IMAGEUPLOAD",
|
||||
"widget": {
|
||||
"name": "upload"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
1,
|
||||
3
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"camera_example_pinhole.jpg",
|
||||
"image"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 21,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
1010.1402587890625,
|
||||
1385.9761962890625
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
262
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Point cloud walker (fly-through video)\n\nSingle image → metric depth → point cloud, then **CameraTrajectoryNode** derives a camera path and **CameraMotionNode** renders a fly-through saved as WEBM.\n\n- **Set:** input image; the move in TransformToMatrix (shift XYZ + theta/phi); frames-per-segment (`n_points`) in CameraMotionNode.\n- **Requires:** Depth-Anything V2 (auto-download).\n- **Note:** the pinhole FOV here is 60° — keep DepthToPointCloud and CameraMotionNode FOVs consistent with your input."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
1,
|
||||
1,
|
||||
0,
|
||||
2,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
3,
|
||||
1,
|
||||
0,
|
||||
3,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
4,
|
||||
2,
|
||||
0,
|
||||
4,
|
||||
0,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
5,
|
||||
5,
|
||||
0,
|
||||
4,
|
||||
1,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
23,
|
||||
4,
|
||||
0,
|
||||
14,
|
||||
0,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
24,
|
||||
14,
|
||||
0,
|
||||
10,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
25,
|
||||
5,
|
||||
0,
|
||||
17,
|
||||
1,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
26,
|
||||
2,
|
||||
0,
|
||||
17,
|
||||
0,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
27,
|
||||
17,
|
||||
0,
|
||||
14,
|
||||
1,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
31,
|
||||
3,
|
||||
0,
|
||||
20,
|
||||
0,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
32,
|
||||
20,
|
||||
0,
|
||||
2,
|
||||
1,
|
||||
"TENSOR"
|
||||
]
|
||||
],
|
||||
"groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"title": "1. Image → point cloud",
|
||||
"bounding": [
|
||||
1550.1402587890625,
|
||||
1325.9761962890625,
|
||||
884.6761474609375,
|
||||
635.9034423828125
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"title": "2. Trajectory",
|
||||
"bounding": [
|
||||
2276.978759765625,
|
||||
1345.860595703125,
|
||||
782.6866455078125,
|
||||
570.35400390625
|
||||
],
|
||||
"color": "#a1309b",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"title": "3. Render & save",
|
||||
"bounding": [
|
||||
2964.30224609375,
|
||||
1426.6483154296875,
|
||||
722.80126953125,
|
||||
566.046875
|
||||
],
|
||||
"color": "#8A8",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
}
|
||||
],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 1.083470594338842,
|
||||
"offset": [
|
||||
-1328.4875281402603,
|
||||
-1380.0248413907814
|
||||
]
|
||||
},
|
||||
"frontendVersion": "1.18.9",
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
@@ -0,0 +1,212 @@
|
||||
{
|
||||
"id": "00000000-0000-0000-0000-000000000000",
|
||||
"revision": 0,
|
||||
"last_node_id": 5,
|
||||
"last_link_id": 3,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 1,
|
||||
"type": "TransformToMatrix",
|
||||
"title": "Start pose (identity)",
|
||||
"pos": [
|
||||
-500,
|
||||
320
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "transformation matrix",
|
||||
"type": "MAT_4X4",
|
||||
"links": [
|
||||
1
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformToMatrix"
|
||||
},
|
||||
"widgets_values": [
|
||||
0.0,
|
||||
0.0,
|
||||
0.0,
|
||||
0.0,
|
||||
0.0
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "TransformToMatrix",
|
||||
"title": "End pose (edit me)",
|
||||
"pos": [
|
||||
-500,
|
||||
540
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "transformation matrix",
|
||||
"type": "MAT_4X4",
|
||||
"links": [
|
||||
2
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformToMatrix"
|
||||
},
|
||||
"widgets_values": [
|
||||
0.0,
|
||||
0.0,
|
||||
0.3,
|
||||
0.0,
|
||||
30.0
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"type": "CameraInterpolationNode",
|
||||
"title": "Interpolate 20 poses",
|
||||
"pos": [
|
||||
-120,
|
||||
430
|
||||
],
|
||||
"size": [
|
||||
226,
|
||||
78
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "initial_matrix",
|
||||
"type": "MAT_4X4",
|
||||
"link": 1
|
||||
},
|
||||
{
|
||||
"name": "final_matrix",
|
||||
"type": "MAT_4X4",
|
||||
"link": 2
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "trajectory",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
3
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "CameraInterpolationNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
20
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "SaveTrajectory",
|
||||
"title": "Save to output/*.npy",
|
||||
"pos": [
|
||||
170,
|
||||
430
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
82
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "trajectory",
|
||||
"type": "TENSOR",
|
||||
"link": 3
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "SaveTrajectory"
|
||||
},
|
||||
"widgets_values": [
|
||||
"ComfyUITrajectory"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
-1080,
|
||||
320
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
430
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Record a camera trajectory\n\nProduces the `.npy` trajectory file consumed by `PC_enricher.json` and `wan_vace_ref_to_video.json` (LoadTrajectory): two poses are SE(3)-interpolated into a smooth 20-step path and saved by **SaveTrajectory** to your ComfyUI **output** directory.\n\n- **Set:** the end pose (shift XYZ in scene units — metric if the cloud came from metric depth — plus theta = pitch, phi = yaw) and `num_steps`.\n- **Then:** move the saved file from `output/` to `input/` so LoadTrajectory can list it. A bundled example (`ComfyUITrajectory_00001.npy`) is already installed by install.py.\n- Chain more CameraInterpolationNode segments (or use CameraTrajectoryNode on a point cloud) for multi-keyframe paths."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
1,
|
||||
1,
|
||||
0,
|
||||
3,
|
||||
0,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
2,
|
||||
2,
|
||||
0,
|
||||
3,
|
||||
1,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
3,
|
||||
3,
|
||||
0,
|
||||
4,
|
||||
0,
|
||||
"TENSOR"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
|
After Width: | Height: | Size: 46 KiB |
@@ -0,0 +1,320 @@
|
||||
{
|
||||
"id": "dd56c0bf-7405-406e-924f-42b2feacb73f",
|
||||
"revision": 0,
|
||||
"last_node_id": 9,
|
||||
"last_link_id": 9,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 3,
|
||||
"type": "SaveWEBM",
|
||||
"pos": [
|
||||
606.5880737304688,
|
||||
1515.5966796875
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
437
|
||||
],
|
||||
"flags": {},
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 5
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"ComfyUI",
|
||||
"vp9",
|
||||
10.000000000000002,
|
||||
32
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 7,
|
||||
"type": "CameraMotionNode",
|
||||
"pos": [
|
||||
176.07933044433594,
|
||||
1504.22705078125
|
||||
],
|
||||
"size": [
|
||||
278.75,
|
||||
270
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"link": 6
|
||||
},
|
||||
{
|
||||
"name": "trajectory",
|
||||
"type": "TENSOR",
|
||||
"link": 9
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "motion_frames",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
5
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "mask_frames",
|
||||
"type": "MASK",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "CameraMotionNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
10,
|
||||
"PINHOLE",
|
||||
90,
|
||||
512,
|
||||
512,
|
||||
1,
|
||||
0,
|
||||
false,
|
||||
false
|
||||
],
|
||||
"title": "Render motion"
|
||||
},
|
||||
{
|
||||
"id": 6,
|
||||
"type": "LoadPointCloud",
|
||||
"pos": [
|
||||
-357.24761962890625,
|
||||
1294.427734375
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
58
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "loaded pointcloud",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
6
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadPointCloud"
|
||||
},
|
||||
"widgets_values": [
|
||||
"ComfyUIPointCloud_00001.ply"
|
||||
],
|
||||
"title": "Load saved cloud (.ply/.npy)"
|
||||
},
|
||||
{
|
||||
"id": 1,
|
||||
"type": "TransformToMatrix",
|
||||
"pos": [
|
||||
-537.2319946289062,
|
||||
1491.40380859375
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "transformation matrix",
|
||||
"type": "MAT_4X4",
|
||||
"links": [
|
||||
7
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformToMatrix"
|
||||
},
|
||||
"widgets_values": [
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0
|
||||
],
|
||||
"title": "Start pose (identity)"
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "TransformToMatrix",
|
||||
"pos": [
|
||||
-531.216796875,
|
||||
1701.1632080078125
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "transformation matrix",
|
||||
"type": "MAT_4X4",
|
||||
"links": [
|
||||
8
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformToMatrix"
|
||||
},
|
||||
"widgets_values": [
|
||||
0.10000000000000002,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0
|
||||
],
|
||||
"title": "End pose (dolly +0.1)"
|
||||
},
|
||||
{
|
||||
"id": 8,
|
||||
"type": "CameraInterpolationNode",
|
||||
"pos": [
|
||||
-121.91971588134766,
|
||||
1597.3514404296875
|
||||
],
|
||||
"size": [
|
||||
200.21640014648438,
|
||||
46
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "initial_matrix",
|
||||
"type": "MAT_4X4",
|
||||
"link": 7
|
||||
},
|
||||
{
|
||||
"name": "final_matrix",
|
||||
"type": "MAT_4X4",
|
||||
"link": 8
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "trajectory",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
9
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "CameraInterpolationNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
2
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 9,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
-1097.2319946289062,
|
||||
1294.427734375
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
236
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Load a saved point cloud & orbit\n\nReloads a point cloud saved by SavePointCloud and renders a short camera move between two poses (identity → 0.1 forward) as a WEBM.\n\n- **Set:** the file in LoadPointCloud (dropdown lists the ComfyUI input dir — run `PointCloud.json` or `fisheye_to_pointcloud.json` first); the two TransformToMatrix poses.\n- **Requires:** nothing beyond this pack."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
5,
|
||||
7,
|
||||
0,
|
||||
3,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
6,
|
||||
6,
|
||||
0,
|
||||
7,
|
||||
0,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
7,
|
||||
1,
|
||||
0,
|
||||
8,
|
||||
0,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
8,
|
||||
5,
|
||||
0,
|
||||
8,
|
||||
1,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
9,
|
||||
8,
|
||||
0,
|
||||
7,
|
||||
1,
|
||||
"TENSOR"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 1.015255979947716,
|
||||
"offset": [
|
||||
636.7531305750655,
|
||||
-1186.3099424359816
|
||||
]
|
||||
},
|
||||
"frontendVersion": "1.21.7",
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
|
After Width: | Height: | Size: 28 KiB |
|
After Width: | Height: | Size: 41 KiB |
@@ -0,0 +1,633 @@
|
||||
"""World-building nodes: depth-scale anchoring, splat world enrichment along a
|
||||
trajectory (render -> outpaint -> SHARP -> align -> fuse) and panorama sphere seeding.
|
||||
|
||||
Contracts implemented here (see SPEC_4D.md):
|
||||
C4: align_depth_scale(new_depth, ref_depth, valid_mask, mode) -> (aligned, scale, shift)
|
||||
|
||||
Heavy dependencies (Flux inpainting / diffusers via OutpaintAnyProjection, SHARP)
|
||||
are only imported/loaded inside methods at call time.
|
||||
"""
|
||||
|
||||
import math
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
import torch
|
||||
from tqdm import tqdm
|
||||
|
||||
try:
|
||||
import folder_paths
|
||||
except ImportError: # Allow notebook usage outside ComfyUI
|
||||
class _FolderPathsStub:
|
||||
def __getattr__(self, name):
|
||||
raise ModuleNotFoundError(
|
||||
"folder_paths is unavailable; this node requires the ComfyUI runtime."
|
||||
)
|
||||
|
||||
folder_paths = _FolderPathsStub()
|
||||
|
||||
try:
|
||||
from . import GS_nodes as _gs
|
||||
except Exception:
|
||||
import GS_nodes as _gs
|
||||
|
||||
GaussianSplats = _gs.GaussianSplats
|
||||
Projection = _gs.Projection
|
||||
DEVICE_CHOICES = _gs.DEVICE_CHOICES
|
||||
_resolve_device_choice = _gs._resolve_device_choice
|
||||
splat_cloud_rotation = _gs.splat_cloud_rotation
|
||||
_stitch_splats = _gs._stitch_splats
|
||||
|
||||
# Zeroth-order real SH constant; rendering with add_sh_bias=True computes
|
||||
# rgb = C0 * f_dc + 0.5, so seeding uses f_dc = (rgb - 0.5) / C0.
|
||||
SH_C0 = 0.28209479177387814
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Lazy accessors for symbols provided by sibling modules / heavy dependencies
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _get_render_gaussians():
|
||||
"""Fetch GS_nodes.render_gaussians (contract C2) with an actionable error."""
|
||||
fn = getattr(_gs, "render_gaussians", None)
|
||||
if fn is None:
|
||||
raise RuntimeError(
|
||||
"GS_nodes.render_gaussians is unavailable. Update GS_nodes.py to a version "
|
||||
"that provides the module-level render_gaussians function (contract C2)."
|
||||
)
|
||||
return fn
|
||||
|
||||
|
||||
def _load_outpaint_node_class():
|
||||
"""Lazy-import OutpaintAnyProjection (pulls in Flux/diffusers machinery)."""
|
||||
try:
|
||||
from .flux_fisheye_filling_nodes import OutpaintAnyProjection
|
||||
return OutpaintAnyProjection
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
from flux_fisheye_filling_nodes import OutpaintAnyProjection
|
||||
return OutpaintAnyProjection
|
||||
except Exception as exc:
|
||||
raise RuntimeError(
|
||||
"OutpaintAnyProjection could not be imported from flux_fisheye_filling_nodes. "
|
||||
"It requires the inpainting_flux custom node package (Flux NF4 inpainting, "
|
||||
"diffusers), which this pack's install.py sets up automatically (ComfyUI-Manager "
|
||||
"runs it on install). Run install.py or fix custom_nodes/inpainting_flux. "
|
||||
f"Import error: {exc}"
|
||||
) from exc
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# C4: robust depth-scale alignment in the disparity domain
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def align_depth_scale(
|
||||
new_depth: torch.Tensor,
|
||||
ref_depth: torch.Tensor,
|
||||
valid_mask: torch.Tensor,
|
||||
mode: str = "scale_shift",
|
||||
) -> Tuple[torch.Tensor, float, float]:
|
||||
"""Least-squares scale(+shift) in DISPARITY (1/d) domain on valid_mask pixels,
|
||||
robust (clip residual outliers, 2 IRLS rounds). Returns (aligned_depth, scale, shift).
|
||||
|
||||
Fits 1/ref_depth ~= scale * (1/new_depth) + shift over valid pixels and returns
|
||||
new_depth remapped through the fitted disparity transform. If the fit is
|
||||
degenerate (too few valid pixels, non-positive/non-finite scale), returns the
|
||||
input depth unchanged with (scale=1.0, shift=0.0).
|
||||
"""
|
||||
if mode not in ("scale", "scale_shift"):
|
||||
raise ValueError(f"Unknown align mode: {mode}")
|
||||
|
||||
nd = torch.as_tensor(new_depth).float()
|
||||
# Harmonize devices: the inputs may arrive on different devices (e.g. a
|
||||
# CUDA motion mask from MotionMaskFromDepth combined with CPU depth
|
||||
# estimates); compute everything on new_depth's device.
|
||||
rd = torch.as_tensor(ref_depth).float().to(nd.device)
|
||||
vm = torch.as_tensor(valid_mask).float().to(nd.device)
|
||||
|
||||
nd_flat = nd.reshape(-1)
|
||||
rd_flat = rd.reshape(-1)
|
||||
if vm.numel() == nd_flat.numel():
|
||||
vm_flat = vm.reshape(-1)
|
||||
else:
|
||||
try:
|
||||
vm_flat = vm.expand_as(nd).reshape(-1)
|
||||
except RuntimeError as exc:
|
||||
raise ValueError(
|
||||
f"valid_mask shape {tuple(vm.shape)} is not broadcastable to depth shape {tuple(nd.shape)}"
|
||||
) from exc
|
||||
|
||||
eps = 1e-8
|
||||
valid = (
|
||||
(vm_flat > 0.5)
|
||||
& (nd_flat > eps)
|
||||
& (rd_flat > eps)
|
||||
& torch.isfinite(nd_flat)
|
||||
& torch.isfinite(rd_flat)
|
||||
)
|
||||
if int(valid.sum().item()) < 10:
|
||||
return nd.clone(), 1.0, 0.0
|
||||
|
||||
x = 1.0 / nd_flat[valid] # new disparity
|
||||
y = 1.0 / rd_flat[valid] # reference disparity
|
||||
w = torch.ones_like(x)
|
||||
|
||||
scale, shift = 1.0, 0.0
|
||||
# Initial weighted LSQ fit + 2 IRLS re-weighting rounds (outlier clipping).
|
||||
for _ in range(3):
|
||||
sw = w.sum().clamp(min=eps)
|
||||
sx = (w * x).sum()
|
||||
sy = (w * y).sum()
|
||||
if mode == "scale_shift":
|
||||
sxx = (w * x * x).sum()
|
||||
sxy = (w * x * y).sum()
|
||||
denom = sw * sxx - sx * sx
|
||||
if float(denom.abs().item()) < eps:
|
||||
s = (sxy / sxx.clamp(min=eps)).item()
|
||||
b = 0.0
|
||||
else:
|
||||
s = float(((sw * sxy - sx * sy) / denom).item())
|
||||
b = float(((sy - s * sx) / sw).item())
|
||||
else:
|
||||
sxx = (w * x * x).sum()
|
||||
sxy = (w * x * y).sum()
|
||||
s = float((sxy / sxx.clamp(min=eps)).item())
|
||||
b = 0.0
|
||||
scale, shift = s, b
|
||||
|
||||
resid = y - (scale * x + shift)
|
||||
sigma = 1.4826 * resid.abs().median()
|
||||
sigma = sigma.clamp(min=eps)
|
||||
w = (resid.abs() <= 2.5 * sigma).float()
|
||||
if float(w.sum().item()) < 10:
|
||||
break
|
||||
|
||||
if not math.isfinite(scale) or scale <= 0.0 or not math.isfinite(shift):
|
||||
return nd.clone(), 1.0, 0.0
|
||||
|
||||
disp = scale / nd.clamp(min=eps) + shift
|
||||
aligned = 1.0 / disp.clamp(min=eps)
|
||||
return aligned, float(scale), float(shift)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Internal helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _coerce_trajectory(trajectory: Any, device: torch.device) -> torch.Tensor:
|
||||
"""Coerce trajectory input to a [K,4,4] float tensor on device."""
|
||||
if isinstance(trajectory, torch.Tensor):
|
||||
traj = trajectory
|
||||
else:
|
||||
traj = torch.as_tensor(trajectory)
|
||||
traj = traj.to(device=device, dtype=torch.float32)
|
||||
if traj.dim() == 2:
|
||||
traj = traj.unsqueeze(0)
|
||||
if traj.dim() != 3 or traj.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"trajectory must be [K,4,4], got shape {tuple(traj.shape)}")
|
||||
return traj
|
||||
|
||||
|
||||
def _project_to_pixels(
|
||||
xyz: torch.Tensor,
|
||||
projection: str,
|
||||
horizontal_fov: float,
|
||||
width: int,
|
||||
height: int,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
"""Project camera-frame points to integer pixel indices.
|
||||
|
||||
Returns (ix [N], iy [N], ray_depth [N], valid [N]) where valid means the point
|
||||
is in front of the camera (pinhole) and lands inside the image bounds. Uses the
|
||||
same projection math as GS_nodes rendering so pixels line up with renders.
|
||||
"""
|
||||
X, Y, Z = xyz.unbind(-1)
|
||||
if projection == "PINHOLE":
|
||||
u, v, depth = _gs._xyz_to_pinhole(X, Y, Z, horizontal_fov)
|
||||
front = Z > 1e-6
|
||||
elif projection == "FISHEYE":
|
||||
u, v, depth = _gs._xyz_to_fisheye(X, Y, Z, horizontal_fov)
|
||||
front = depth > 1e-6
|
||||
else:
|
||||
u, v, depth = _gs._xyz_to_equirect(X, Y, Z, horizontal_fov)
|
||||
front = depth > 1e-6
|
||||
|
||||
ix = torch.round((u * 0.5 + 0.5) * (width - 1)).long()
|
||||
iy = torch.round((v * 0.5 + 0.5) * (height - 1)).long()
|
||||
inside = (u >= -1.0) & (u <= 1.0) & (v >= -1.0) & (v <= 1.0)
|
||||
valid = front & inside & torch.isfinite(u) & torch.isfinite(v)
|
||||
ix = ix.clamp(0, width - 1)
|
||||
iy = iy.clamp(0, height - 1)
|
||||
return ix, iy, depth, valid
|
||||
|
||||
|
||||
def _pad_f_rest_to_order(splats: GaussianSplats, sh_order: int) -> GaussianSplats:
|
||||
"""Zero-pad SH coefficients so splats match the requested (higher) SH order.
|
||||
|
||||
Delegates to GS_nodes._pad_sh_order, which handles the renderer's
|
||||
channel-major SH layout (cat([f_dc, f_rest]).view(-1, 3, total)) correctly.
|
||||
Naively appending zeros to f_rest would shift the green/blue DC terms into
|
||||
the red channel's l>=1 slots and corrupt colors.
|
||||
"""
|
||||
return _gs._pad_sh_order(splats, sh_order)
|
||||
|
||||
|
||||
def _match_sh_orders(a: GaussianSplats, b: GaussianSplats) -> Tuple[GaussianSplats, GaussianSplats]:
|
||||
"""Bring two splat sets to a common (max) SH order via zero padding."""
|
||||
return _gs._match_sh_orders(a, b)
|
||||
|
||||
|
||||
def _scale_splats_metric(splats: GaussianSplats, factor: float) -> GaussianSplats:
|
||||
"""Uniformly rescale splat positions and sizes by a metric factor."""
|
||||
out = splats.clone()
|
||||
out.xyz = out.xyz * factor
|
||||
out.scale = out.scale + math.log(max(factor, 1e-12))
|
||||
return out
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Nodes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class DepthScaleAnchor:
|
||||
"""Aligns a depth map's scale (and optionally shift) to a reference depth map
|
||||
using a robust least-squares fit in the disparity domain (contract C4)."""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
"new_depth": ("TENSOR", {"tooltip": "Depth map to be aligned (any shape)."}),
|
||||
"ref_depth": ("TENSOR", {"tooltip": "Reference metric depth map (same shape)."}),
|
||||
"valid_mask": ("MASK", {"tooltip": "1.0 where both depths are trustworthy."}),
|
||||
"mode": (
|
||||
["scale", "scale_shift"],
|
||||
{"default": "scale_shift", "tooltip": "Fit scale only, or scale + shift, in disparity (1/d) domain."},
|
||||
),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("TENSOR", "FLOAT", "FLOAT")
|
||||
RETURN_NAMES = ("aligned_depth", "scale", "shift")
|
||||
FUNCTION = "anchor"
|
||||
CATEGORY = "Camera/World"
|
||||
DESCRIPTION = "Robustly aligns a depth map to a reference depth via disparity-domain scale(+shift)."
|
||||
|
||||
def anchor(
|
||||
self,
|
||||
new_depth: torch.Tensor,
|
||||
ref_depth: torch.Tensor,
|
||||
valid_mask: torch.Tensor,
|
||||
mode: str = "scale_shift",
|
||||
):
|
||||
aligned, scale, shift = align_depth_scale(new_depth, ref_depth, valid_mask, mode=mode)
|
||||
return (aligned, scale, shift)
|
||||
|
||||
|
||||
class SplatTrajectoryEnricher:
|
||||
"""World-expansion loop for Gaussian splats.
|
||||
|
||||
For each pose along a trajectory: render the current splats, detect uncovered
|
||||
(hole) regions, fill them with Flux outpainting, lift the filled view to new
|
||||
splats with SHARP, align the SHARP metric scale to the rendered reference
|
||||
depth, keep only the splats that cover holes, transform them to world space
|
||||
and fuse them into the running splat set.
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
choices = _gs._list_sharp_checkpoint_choices()
|
||||
return {
|
||||
"required": {
|
||||
"splats": ("GSPLAT",),
|
||||
"trajectory": ("TENSOR", {"tooltip": "[K,4,4] world-to-camera matrices of poses to visit."}),
|
||||
"camera_projection": (Projection.PROJECTIONS, {}),
|
||||
"horizontal_fov": ("FLOAT", {"default": 90.0, "min": 1.0, "max": 360.0}),
|
||||
"width": ("INT", {"default": 512, "min": 8, "max": 8192}),
|
||||
"height": ("INT", {"default": 512, "min": 8, "max": 8192}),
|
||||
"checkpoint": (
|
||||
choices,
|
||||
{
|
||||
"default": _gs._SHARP_DEFAULT_CHECKPOINT_LABEL,
|
||||
"file_chooser": True,
|
||||
"tooltip": "SHARP .pt checkpoint from the input folder, or download the default model.",
|
||||
},
|
||||
),
|
||||
"prompt": ("STRING", {"default": "", "multiline": True}),
|
||||
"num_inference_steps": ("INT", {"default": 28, "min": 10, "max": 60}),
|
||||
"guidance_scale": ("FLOAT", {"default": 5.0, "min": 0.1, "max": 30.0}),
|
||||
"mask_blur": ("INT", {"default": 5, "min": 0, "max": 512}),
|
||||
"hole_min_frac": (
|
||||
"FLOAT",
|
||||
{"default": 0.02, "min": 0.0, "max": 1.0, "step": 0.001,
|
||||
"tooltip": "Skip a view if the uncovered area is below this fraction of pixels."},
|
||||
),
|
||||
"stitch_voxel_size": ("FLOAT", {"default": 0.01, "min": 0.0, "max": 10.0}),
|
||||
"max_views": ("INT", {"default": 10, "min": 1, "max": 1000}),
|
||||
},
|
||||
"optional": {
|
||||
"device": (DEVICE_CHOICES, {"default": "auto"}),
|
||||
"cache_flux": (
|
||||
"BOOLEAN",
|
||||
{"default": True,
|
||||
"tooltip": "Keep the Flux inpainting pipeline loaded between views (avoids a multi-GB "
|
||||
"model reload per view). Disable to free VRAM after each outpaint on "
|
||||
"low-memory GPUs."},
|
||||
),
|
||||
"patch_projection": (Projection.PROJECTIONS, {"default": "PINHOLE", "tooltip": "Projection used for the outpaint patch."}),
|
||||
"patch_horiz_fov": ("FLOAT", {"default": 90.0, "min": 1.0, "max": 180.0}),
|
||||
"patch_res": ("INT", {"default": 1024, "min": 64, "max": 8192}),
|
||||
"patch_phi": ("FLOAT", {"default": 0.0, "min": -180.0, "max": 180.0}),
|
||||
"patch_theta": ("FLOAT", {"default": 0.0, "min": -90.0, "max": 90.0}),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("GSPLAT", "IMAGE", "IMAGE")
|
||||
RETURN_NAMES = ("enriched_splats", "last_render", "last_filled")
|
||||
FUNCTION = "enrich"
|
||||
CATEGORY = "Camera/World"
|
||||
DESCRIPTION = (
|
||||
"Expands a splat world along a camera trajectory: render, outpaint holes with Flux, "
|
||||
"lift with SHARP, scale-align, and smart-stitch the new content."
|
||||
)
|
||||
|
||||
@torch.no_grad()
|
||||
def enrich(
|
||||
self,
|
||||
splats: GaussianSplats,
|
||||
trajectory: torch.Tensor,
|
||||
camera_projection: str,
|
||||
horizontal_fov: float,
|
||||
width: int,
|
||||
height: int,
|
||||
checkpoint: str,
|
||||
prompt: str,
|
||||
num_inference_steps: int,
|
||||
guidance_scale: float,
|
||||
mask_blur: int,
|
||||
hole_min_frac: float,
|
||||
stitch_voxel_size: float,
|
||||
max_views: int,
|
||||
device: str = "auto",
|
||||
cache_flux: bool = True,
|
||||
patch_projection: str = "PINHOLE",
|
||||
patch_horiz_fov: float = 90.0,
|
||||
patch_res: int = 1024,
|
||||
patch_phi: float = 0.0,
|
||||
patch_theta: float = 0.0,
|
||||
) -> Tuple[GaussianSplats, torch.Tensor, torch.Tensor]:
|
||||
# Fail fast: the SHARP lift (ImageToSplat) is pinhole-only and requires
|
||||
# horizontal_fov < 179 degrees. Validating here avoids crashing in the
|
||||
# lift step AFTER minutes of rendering + Flux outpainting work.
|
||||
if not (0.0 < float(horizontal_fov) < 179.0):
|
||||
raise ValueError(
|
||||
"SplatTrajectoryEnricher lifts filled views with SHARP (pinhole), which requires "
|
||||
f"0 < horizontal_fov < 179 degrees (got {horizontal_fov}). For panoramic worlds "
|
||||
"(EQUIRECTANGULAR/FISHEYE with fov >= 179), visit several narrower pinhole poses "
|
||||
"along the trajectory instead (e.g. 90-120 degree views after SphereSplatSeed)."
|
||||
)
|
||||
render_gaussians = _get_render_gaussians()
|
||||
outpaint_cls = _load_outpaint_node_class()
|
||||
outpaint_node = outpaint_cls()
|
||||
image_to_splat = _gs.ImageToSplat()
|
||||
|
||||
target_device = _resolve_device_choice(device)
|
||||
current = splats.to(target_device) if splats.xyz.device != target_device else splats
|
||||
traj = _coerce_trajectory(trajectory, target_device)
|
||||
|
||||
if camera_projection != "PINHOLE":
|
||||
print(
|
||||
"[SplatTrajectoryEnricher] Warning: SHARP assumes pinhole geometry; "
|
||||
f"lifting filled {camera_projection} views may distort new splats."
|
||||
)
|
||||
|
||||
last_render = torch.zeros((1, height, width, 3), device=target_device)
|
||||
last_filled = torch.zeros((1, height, width, 3), device=target_device)
|
||||
added_views = 0
|
||||
|
||||
for pose in tqdm(traj[: max(1, int(max_views))], desc="Enriching splat world"):
|
||||
# 1) Render the current world from this pose.
|
||||
image, alpha, disparity = render_gaussians(
|
||||
current,
|
||||
pose,
|
||||
camera_projection,
|
||||
horizontal_fov,
|
||||
width,
|
||||
height,
|
||||
max_splats=0,
|
||||
opacity_is_logit=True,
|
||||
add_sh_bias=True,
|
||||
render_mode="auto",
|
||||
device=str(target_device).split(":")[0],
|
||||
)
|
||||
last_render = image
|
||||
|
||||
alpha_map = alpha.view(height, width).to(target_device)
|
||||
disp_map = disparity.view(height, width).to(target_device)
|
||||
hole_mask = (alpha_map < 0.5).float()
|
||||
|
||||
hole_frac = float(hole_mask.mean().item())
|
||||
if hole_frac < hole_min_frac:
|
||||
continue
|
||||
|
||||
# 2) Outpaint the uncovered region.
|
||||
filled_img, _ = outpaint_node.outpaint_any(
|
||||
image,
|
||||
input_projection=camera_projection,
|
||||
input_horiz_fov=horizontal_fov,
|
||||
output_projection=camera_projection,
|
||||
output_horiz_fov=horizontal_fov,
|
||||
output_width=width,
|
||||
output_height=height,
|
||||
patch_projection=patch_projection,
|
||||
patch_horiz_fov=patch_horiz_fov,
|
||||
patch_res=patch_res,
|
||||
patch_phi=patch_phi,
|
||||
patch_theta=patch_theta,
|
||||
prompt=prompt,
|
||||
num_inference_steps=num_inference_steps,
|
||||
# cached=True keeps the Flux NF4 pipeline resident between views
|
||||
# (cached=False forced a full multi-GB pipeline reload per view).
|
||||
cached=bool(cache_flux),
|
||||
guidance_scale=guidance_scale,
|
||||
mask_blur=mask_blur,
|
||||
mask=hole_mask.unsqueeze(0),
|
||||
debug=False,
|
||||
)
|
||||
last_filled = filled_img
|
||||
|
||||
# 3) Lift the filled view to splats in this camera frame (SHARP, metric).
|
||||
new_splats, = image_to_splat.image_to_splat(
|
||||
filled_img,
|
||||
horizontal_fov,
|
||||
checkpoint,
|
||||
device,
|
||||
)
|
||||
new_splats = new_splats.to(target_device)
|
||||
if len(new_splats) == 0:
|
||||
continue
|
||||
|
||||
# 4) Robust metric-scale alignment against the rendered reference depth.
|
||||
# Reference ray depth from the renderer: disparity = alpha / depth.
|
||||
ix, iy, sharp_depth, proj_valid = _project_to_pixels(
|
||||
new_splats.xyz, camera_projection, horizontal_fov, width, height
|
||||
)
|
||||
samp_alpha = alpha_map[iy, ix]
|
||||
samp_disp = disp_map[iy, ix]
|
||||
overlap = proj_valid & (samp_alpha >= 0.5) & (samp_disp > 1e-6) & (sharp_depth > 1e-6)
|
||||
if int(overlap.sum().item()) >= 10:
|
||||
d_ref = (samp_alpha[overlap] / samp_disp[overlap]).clamp(min=1e-6)
|
||||
ratio = d_ref / sharp_depth[overlap]
|
||||
scale_factor = float(ratio.median().item())
|
||||
if math.isfinite(scale_factor) and scale_factor > 0.0:
|
||||
new_splats = _scale_splats_metric(new_splats, scale_factor)
|
||||
|
||||
# 5) Keep only NEW content: splats whose projected pixel lies in a hole.
|
||||
samp_hole = hole_mask[iy, ix]
|
||||
keep = proj_valid & (samp_hole > 0.5)
|
||||
if not bool(keep.any().item()):
|
||||
continue
|
||||
new_splats = new_splats[keep]
|
||||
|
||||
# 6) Camera frame -> world frame (pose is world-to-camera).
|
||||
new_world = splat_cloud_rotation(new_splats, torch.inverse(pose))
|
||||
|
||||
# 7) Fuse into the running world. Concatenation is cheap; the full
|
||||
# smart voxel reduce is deferred to a single pass after the loop,
|
||||
# so each view does not re-copy and re-unique-sort the entire
|
||||
# accumulated cloud (O(views x N) work/memory otherwise).
|
||||
cur_m, new_m = _match_sh_orders(current, new_world)
|
||||
current = _gs._concat_splats([cur_m, new_m])
|
||||
added_views += 1
|
||||
|
||||
if added_views > 0 and stitch_voxel_size > 0.0:
|
||||
current = _stitch_splats([current], "smart", stitch_voxel_size, 5.0)
|
||||
|
||||
return (current, last_render, last_filled)
|
||||
|
||||
|
||||
class SphereSplatSeed:
|
||||
"""Seeds a 360-degree splat world from an equirectangular panorama: one Gaussian
|
||||
per (subsampled) pixel, placed on a depth sphere around the origin."""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
"image": ("IMAGE", {"tooltip": "Equirectangular panorama [1,H,W,3]."}),
|
||||
"horizontal_fov": ("FLOAT", {"default": 360.0, "min": 1.0, "max": 360.0}),
|
||||
"radius": ("FLOAT", {"default": 5.0, "min": 0.01, "max": 10000.0, "tooltip": "Sphere radius used when no depth map is provided."}),
|
||||
"splat_scale_frac": (
|
||||
"FLOAT",
|
||||
{"default": 1.5, "min": 0.1, "max": 10.0,
|
||||
"tooltip": "Splat sigma as a fraction of the local point spacing (larger = smoother, fewer holes)."},
|
||||
),
|
||||
"stride": ("INT", {"default": 2, "min": 1, "max": 64, "tooltip": "Pixel subsampling stride (1 Gaussian per stride x stride block)."}),
|
||||
},
|
||||
"optional": {
|
||||
"depth": ("TENSOR", {"tooltip": "Optional ray-depth map [H,W] (or [1,H,W]/[H,W,1]) matching the panorama."}),
|
||||
"opacity_logit": ("FLOAT", {"default": 6.0, "min": -10.0, "max": 20.0}),
|
||||
"device": (DEVICE_CHOICES, {"default": "auto"}),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("GSPLAT",)
|
||||
RETURN_NAMES = ("splats",)
|
||||
FUNCTION = "seed_sphere"
|
||||
CATEGORY = "Camera/World"
|
||||
DESCRIPTION = "Converts an equirectangular panorama into a Gaussian sphere seeding a 360-degree world."
|
||||
|
||||
@torch.no_grad()
|
||||
def seed_sphere(
|
||||
self,
|
||||
image: torch.Tensor,
|
||||
horizontal_fov: float = 360.0,
|
||||
radius: float = 5.0,
|
||||
splat_scale_frac: float = 1.5,
|
||||
stride: int = 2,
|
||||
depth: Optional[torch.Tensor] = None,
|
||||
opacity_logit: float = 6.0,
|
||||
device: str = "auto",
|
||||
) -> Tuple[GaussianSplats]:
|
||||
target_device = _resolve_device_choice(device)
|
||||
|
||||
img = image
|
||||
if img.dim() == 4:
|
||||
img = img[0]
|
||||
if img.dim() != 3 or img.shape[-1] < 3:
|
||||
raise ValueError(f"Expected IMAGE [1,H,W,3], got shape {tuple(image.shape)}")
|
||||
img = img[..., :3].to(device=target_device, dtype=torch.float32)
|
||||
H, W = int(img.shape[0]), int(img.shape[1])
|
||||
|
||||
depth_map = None
|
||||
if depth is not None:
|
||||
d = torch.as_tensor(depth).to(device=target_device, dtype=torch.float32)
|
||||
if d.dim() == 3:
|
||||
# [1,H,W], [T,H,W] (take first) or [H,W,1]
|
||||
d = d[..., 0] if d.shape[-1] == 1 else d[0]
|
||||
if d.dim() != 2:
|
||||
raise ValueError(f"depth must reduce to [H,W], got shape {tuple(depth.shape)}")
|
||||
if d.shape != (H, W):
|
||||
d = torch.nn.functional.interpolate(
|
||||
d.unsqueeze(0).unsqueeze(0), size=(H, W), mode="bilinear", align_corners=True
|
||||
)[0, 0]
|
||||
depth_map = d.clamp(min=1e-6)
|
||||
|
||||
stride = max(1, int(stride))
|
||||
ys = torch.arange(0, H, stride, device=target_device)
|
||||
xs = torch.arange(0, W, stride, device=target_device)
|
||||
yy, xx = torch.meshgrid(ys, xs, indexing="ij")
|
||||
yy = yy.reshape(-1)
|
||||
xx = xx.reshape(-1)
|
||||
|
||||
# Match the renderer's equirect mapping (GS_nodes._xyz_to_equirect):
|
||||
# u = lon / (fov_rad/2), v = lat / (pi/2), px = (u*0.5+0.5)*(W-1)
|
||||
fov_rad = math.radians(horizontal_fov)
|
||||
u = xx.float() / max(W - 1, 1) * 2.0 - 1.0
|
||||
v = yy.float() / max(H - 1, 1) * 2.0 - 1.0
|
||||
lon = u * (fov_rad / 2.0)
|
||||
lat = v * (math.pi / 2.0)
|
||||
|
||||
if depth_map is not None:
|
||||
d = depth_map[yy, xx]
|
||||
else:
|
||||
d = torch.full_like(lon, float(radius))
|
||||
|
||||
cos_lat = torch.cos(lat)
|
||||
X = d * cos_lat * torch.sin(lon)
|
||||
Y = d * torch.sin(lat)
|
||||
Z = d * cos_lat * torch.cos(lon)
|
||||
xyz = torch.stack([X, Y, Z], dim=-1)
|
||||
|
||||
rgb = img[yy, xx, :]
|
||||
# Rendering with add_sh_bias=True evaluates rgb = C0 * f_dc + 0.5.
|
||||
f_dc = (rgb - 0.5) / SH_C0
|
||||
|
||||
# Isotropic sigma from local angular spacing (radians per sample) times depth.
|
||||
ang_spacing = float(stride) * max(fov_rad / max(W, 1), math.pi / max(H, 1))
|
||||
sigma = (splat_scale_frac * ang_spacing * d).clamp(min=1e-6)
|
||||
scale = torch.log(sigma).unsqueeze(-1).expand(-1, 3).contiguous()
|
||||
|
||||
n = xyz.shape[0]
|
||||
rotation = torch.zeros((n, 4), device=target_device, dtype=torch.float32)
|
||||
rotation[:, 0] = 1.0 # identity wxyz quaternion
|
||||
opacity = torch.full((n, 1), float(opacity_logit), device=target_device, dtype=torch.float32)
|
||||
f_rest = torch.zeros((n, 0), device=target_device, dtype=torch.float32)
|
||||
|
||||
splats = GaussianSplats(
|
||||
xyz=xyz,
|
||||
scale=scale,
|
||||
rotation=rotation,
|
||||
opacity=opacity,
|
||||
f_dc=f_dc,
|
||||
f_rest=f_rest,
|
||||
sh_order=0,
|
||||
)
|
||||
return (splats,)
|
||||
|
||||
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"DepthScaleAnchor": DepthScaleAnchor,
|
||||
"SplatTrajectoryEnricher": SplatTrajectoryEnricher,
|
||||
"SphereSplatSeed": SphereSplatSeed,
|
||||
}
|
||||