Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a2f48da1a4 | ||
|
|
012ab737d6 | ||
|
|
2f9aa76478 | ||
|
|
918c882abd | ||
|
|
79a1c2d6f3 | ||
|
|
d77af07189 | ||
|
|
6f8ed9d382 | ||
|
|
d85c2125d1 | ||
|
|
29a39aa7fd | ||
|
|
358b2b55ae | ||
|
|
8598a7b51c | ||
|
|
90df387c2a | ||
|
|
74d71a91f2 | ||
|
|
55820b53e8 | ||
|
|
f6b6707dd8 | ||
|
|
49f9700880 | ||
|
|
b961c2988d | ||
|
|
2b1b716495 | ||
|
|
839fa5883a | ||
|
|
74f98be5c2 | ||
|
|
c1e1b55464 | ||
|
|
08b48b2fcb | ||
|
|
4812973e0f | ||
|
|
57fc1e8593 | ||
|
|
abdfdc188e | ||
|
|
ca6f904fc3 | ||
|
|
696a4ad763 | ||
|
|
d67cd61c3c | ||
|
|
17a0edb3c1 | ||
|
|
5592382b44 | ||
|
|
bc358f8263 |
@@ -0,0 +1,10 @@
|
||||
# Excluded from the ComfyUI Registry archive (not from git).
|
||||
demo_images/
|
||||
notebooks/
|
||||
docs/
|
||||
screenshot1.ply
|
||||
__pycache__/
|
||||
models/
|
||||
.github/
|
||||
Makefile
|
||||
install.sh
|
||||
@@ -0,0 +1,28 @@
|
||||
name: Publish to Comfy registry
|
||||
on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- "pyproject.toml"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
publish-node:
|
||||
name: Publish Custom Node to registry
|
||||
runs-on: ubuntu-latest
|
||||
if: ${{ github.repository_owner == 'Alexankharin' }}
|
||||
steps:
|
||||
- name: Check out code
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
# The SHARP submodule must be materialized so [tool.comfy].includes
|
||||
# can pack it into the published archive.
|
||||
submodules: recursive
|
||||
- name: Publish Custom Node
|
||||
uses: Comfy-Org/publish-node-action@main
|
||||
with:
|
||||
personal_access_token: ${{ secrets.REGISTRY_ACCESS_TOKEN }}
|
||||
@@ -0,0 +1,34 @@
|
||||
name: Validate workflows & smoke tests
|
||||
|
||||
# Guards against node-schema drift silently breaking the shipped workflows:
|
||||
# validate_workflows.py compares every stored widgets_values against the
|
||||
# current INPUT_TYPES (see docs/workflows_review.md).
|
||||
on:
|
||||
pull_request:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
validate:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check out code
|
||||
uses: actions/checkout@v4
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
- name: Install CPU test dependencies
|
||||
run: |
|
||||
pip install torch torchvision --index-url https://download.pytorch.org/whl/cpu
|
||||
pip install numpy pillow scipy tqdm
|
||||
- name: Workflow schema validation
|
||||
run: python notebooks/validate_workflows.py
|
||||
- name: Installer logic tests
|
||||
run: python notebooks/test_install_logic.py
|
||||
- name: 4D node smoke tests
|
||||
run: python notebooks/smoke_test_4d.py
|
||||
@@ -0,0 +1,2 @@
|
||||
__pycache__/
|
||||
*.pyc
|
||||
@@ -0,0 +1,3 @@
|
||||
[submodule "submodules/ml-sharpt"]
|
||||
path = submodules/ml-sharpt
|
||||
url = https://github.com/apple/ml-sharp
|
||||
@@ -0,0 +1,29 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Alexander Kharin
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
---
|
||||
|
||||
Note: the bundled directory `submodules/ml-sharpt` contains Apple's ml-sharp
|
||||
project and is licensed separately under the terms in
|
||||
`submodules/ml-sharpt/LICENSE` (source) and `submodules/ml-sharpt/LICENSE_MODEL`
|
||||
(model weights, research-only). The MIT license above does not apply to that
|
||||
directory.
|
||||
@@ -1,6 +1,7 @@
|
||||
# camera-comfyUI
|
||||
[](https://deepwiki.com/Alexankharin/camera-comfyUI)
|
||||

|
||||

|
||||
|
||||
> Custom ComfyUI nodes for advanced reprojections, point cloud processing, and camera-driven workflows.
|
||||
|
||||
@@ -13,6 +14,7 @@
|
||||
* [Installation](#installation)
|
||||
* [Node Categories](#node-categories)
|
||||
* [Node Reference](#node-reference)
|
||||
* [Video → 4D World](#video--4d-world)
|
||||
* [Workflows](#workflows)
|
||||
* [Example Workflows](#example-workflows)
|
||||
* [Contributing](#contributing)
|
||||
@@ -34,6 +36,24 @@ A collection of ComfyUI custom nodes to handle diverse camera projections (pinho
|
||||
|
||||
## Installation
|
||||
|
||||
### Option A — ComfyUI Manager (recommended)
|
||||
|
||||
The node pack is published to the [ComfyUI Registry](https://registry.comfy.org) as **`camera-comfyui`** (publisher `alexk`). In ComfyUI, open **Manager → Custom Nodes Manager**, search for **camera-comfyUI**, and click **Install**, then restart ComfyUI.
|
||||
|
||||
Installation is fully automatic: ComfyUI-Manager installs `requirements.txt` and then runs this pack's `install.py`, which sets up everything the optional nodes need — no manual steps:
|
||||
|
||||
* **vggt** (`VideoPoseEstimator`) — pip-installed from GitHub over https (it is not on PyPI).
|
||||
* **SHARP** (`ImageToSplat`, `VideoToFusedSplats`, …) — the `submodules/ml-sharpt` checkout is bundled in the registry package (and fetched via `git submodule`/clone for git installs), and its Python deps come from `requirements.txt`.
|
||||
* **gsplat** — pip-installed; its CUDA kernels JIT-compile on first use.
|
||||
* **ComfyUI-Flux-Inpainting** (`OutpaintAnyProjection`, `SplatTrajectoryEnricher`) — cloned automatically into `custom_nodes/inpainting_flux` (skipped if you already have the pack under any of its usual folder names).
|
||||
* **Example inputs** — the sample image/trajectory files referenced by the bundled workflows are copied into ComfyUI's `input/` folder, so the templates run immediately.
|
||||
|
||||
Each step is optional and non-fatal: if one fails (e.g. no network), only the nodes that need it stay disabled — re-run `python install.py` inside the pack folder to retry.
|
||||
|
||||
> **Maintainers:** releases are automated — bumping `version` in `pyproject.toml` on `main` triggers `.github/workflows/publish_action.yml`, which publishes the new version to the registry (requires the `REGISTRY_ACCESS_TOKEN` repo secret).
|
||||
|
||||
### Option B — Manual install (git)
|
||||
|
||||
1. **Clone** into your ComfyUI custom nodes folder:
|
||||
|
||||
```bash
|
||||
@@ -46,28 +66,29 @@ A collection of ComfyUI custom nodes to handle diverse camera projections (pinho
|
||||
sudo apt-get update && sudo apt-get install build-essential ffmpeg libsm6 libxext6 -y
|
||||
```
|
||||
|
||||
3. **Python Requirements**:
|
||||
3. **Python Requirements + optional dependencies** — one command sets up everything (base requirements, vggt, the SHARP submodule, gsplat, and the `inpainting_flux` sibling pack):
|
||||
|
||||
```bash
|
||||
pip install -r custom_nodes/camera-comfyUI/requirements.txt
|
||||
cd custom_nodes/camera-comfyUI && python install.py
|
||||
```
|
||||
|
||||
* *Optional:* `open3d` for GUI point cloud tools.
|
||||
This is the same script ComfyUI-Manager runs automatically; it is idempotent, and every optional step is non-fatal.
|
||||
|
||||
4. **Additional Nodes** (for certain workflows):
|
||||
**What it covers** (for reference — no manual action needed):
|
||||
|
||||
* Clone the following repositories directly into your `custom_nodes` folder:
|
||||
* [ComfyUI-Flux-Inpainting](https://github.com/rubi-du/ComfyUI-Flux-Inpainting)
|
||||
* [ComfyUI-Image-Filters](https://github.com/spacepxl/ComfyUI-Image-Filters)
|
||||
* **Important:** If the `ComfyUI-Flux-Inpainting` repository is cloned as `ComfyUI-Flux-Inpainting-main`, rename the folder to `inpainting_flux`:
|
||||
```bash
|
||||
mv custom_nodes/ComfyUI-Flux-Inpainting-main custom_nodes/inpainting_flux
|
||||
```
|
||||
* **gsplat** — CUDA-accelerated Gaussian splat rasterizer. Required by `SplatPolish`, SHARP, and the fast render backend for `RenderSplat` / `RenderSplats4D*`. Kernels JIT-compile on first use (needs a CUDA GPU + matching PyTorch build).
|
||||
* **vggt** — camera pose + depth estimation (`VideoPoseEstimator`). Not on PyPI — installed with `pip install git+https://github.com/facebookresearch/vggt.git`. A sibling clone of [facebookresearch/vggt](https://github.com/facebookresearch/vggt) in your ComfyUI root also works. The `facebook/VGGT-1B` weights (~5 GB) download via `huggingface_hub` on first use.
|
||||
* **CoTracker3** — point tracking for `EstimateTracks`. Fetched automatically via `torch.hub` on first use.
|
||||
* **SHARP** — image→splat prediction (`ImageToSplat`, `FisheyeToGaussian`, `VideoToFusedSplats`, `SplatTrajectoryEnricher`). Lives as the git submodule at `submodules/ml-sharpt` ([apple/ml-sharp](https://github.com/apple/ml-sharp)); `install.py` initializes it for you.
|
||||
* **ComfyUI-Flux-Inpainting** — cloned into `custom_nodes/inpainting_flux` if missing. Any of the usual folder names (`inpainting_flux`, `ComfyUI-Flux-Inpainting`, `ComfyUI-Flux-Inpainting-main`) is detected — no renaming needed.
|
||||
|
||||
5. **Flux Models** (Hugging Face):
|
||||
4. **Additional Nodes** (only for some example workflows):
|
||||
|
||||
* [ComfyUI-Image-Filters](https://github.com/spacepxl/ComfyUI-Image-Filters) — install via Manager or clone into `custom_nodes`.
|
||||
|
||||
5. **Flux Models** (Hugging Face, only for gated models):
|
||||
|
||||
```bash
|
||||
pip install huggingface_hub
|
||||
huggingface-cli login
|
||||
```
|
||||
|
||||
@@ -93,13 +114,33 @@ A collection of ComfyUI custom nodes to handle diverse camera projections (pinho
|
||||
* ### Point Cloud Nodes
|
||||
|
||||
* `DepthToPointCloud`, `TransformPointCloud`, `ProjectPointCloud`, `PointCloudUnion`
|
||||
* `PointCloudCleaner`, `LoadPointCloud`, `SavePointCloud`, `ProjectAndClean`
|
||||
* `PointCloudCleaner`, `LoadPointCloud`, `SavePointCloud`, `ProjectAndClean`, `DepthEdgeFilter`
|
||||
|
||||
* ### Trajectory Nodes
|
||||
|
||||
* `CameraMotionNode`, `CameraInterpolationNode`, `CameraTrajectoryNode`
|
||||
* `SaveTrajectory`, `LoadTrajectory`, `PointcloudTrajectoryEnricher`
|
||||
|
||||
* ### Gaussian Splat Nodes
|
||||
|
||||
* `LoadPlySplat`, `SavePlySplat`, `ImageToSplat`, `FisheyeToGaussian`
|
||||
* `RotateSplats`, `MergeSplats`, `FuseSplats`, `RenderSplat`
|
||||
* `VideoToFusedSplats`, `SplatPolish`
|
||||
|
||||
* ### 4D Gaussian Splat Nodes
|
||||
|
||||
* `MotionMaskFromDepth`, `EstimateTracks`, `TracksToTrajectories`, `SplitSplatsByMask`
|
||||
* `BuildSplats4D`, `RenderSplats4DFrame`, `RenderSplats4DVideo`
|
||||
* `SaveSplats4D`, `LoadSplats4D`
|
||||
|
||||
* ### Pose Nodes
|
||||
|
||||
* `VideoPoseEstimator`, `TrajectoryInvert`, `TrajectoryCompose`
|
||||
|
||||
* ### World Nodes
|
||||
|
||||
* `DepthScaleAnchor`, `SplatTrajectoryEnricher`, `SphereSplatSeed`
|
||||
|
||||
---
|
||||
|
||||
## Node Reference
|
||||
@@ -115,31 +156,97 @@ A collection of ComfyUI custom nodes to handle diverse camera projections (pinho
|
||||
| `DepthToPointCloud` | Converts Depth and image to → 3D point cloud tensor (N×7). |
|
||||
| `DepthToImageNode` | Converts depth to image (N×3) using a color map. |
|
||||
| `ZDepthToRayDepthNode` | Converts Z-depth (output of metric-depth-anything) to ray depth to compensate lens curvature. |
|
||||
| `TransformPointCloud` | Applies 4×4 rotation matrix to point cloud |
|
||||
| `TransformPointCloud` | Applies 4×4 rotation matrix to point cloud. |
|
||||
| `ProjectPointCloud` | Z-buffer–based projection of point cloud into image + mask. |
|
||||
| `PointCloudCleaner` | Removes isolated points via voxel filtering. |
|
||||
| `PointCloudUnion` | Combines multiple point clouds into one. |
|
||||
| `LoadPointCloud` | Loads a point cloud from `.npy` or `.ply` format. |
|
||||
| `SavePointCloud` | Saves a point cloud to `.npy` or `.ply` format. |
|
||||
| `CameraMotionNode` | Generates image and mask sequences along a camera trajectory with optional mask dilation/inversion. |
|
||||
| `CameraInterpolationNode` | Builds a trajectory tensor from two poses. |
|
||||
| `CameraTrajectoryNode` | Interactive Open3D GUI for recording camera waypoints. |
|
||||
| `PointCloudCleaner` | Removes isolated points via voxel filtering. |
|
||||
| `SaveTrajectory` | Saves a trajectory tensor to a file. |
|
||||
| `LoadTrajectory` | Loads a trajectory tensor from a file. |
|
||||
| `VideoCameraMotionSequence` | Processes video frames and depth maps along a camera trajectory, generating reprojected outputs. |
|
||||
| `DepthFramesToVideo` | Converts a sequence of depth maps into video frame tensors for saving. |
|
||||
| `VideoMetricDepthEstimate` | Estimates metric depth for a sequence of frames using VideoDepthAnything. |
|
||||
| `DepthEdgeFilter` | Detects "flying pixel" depth discontinuities and outputs a validity mask (1.0 = valid). |
|
||||
| `LoadPlySplat` | Loads a 3D Gaussian Splatting `.ply` file into a `GSPLAT` object. |
|
||||
| `SavePlySplat` | Saves a `GSPLAT` to the ComfyUI output directory as a `.ply` file. |
|
||||
| `ImageToSplat` | Predicts Gaussian splats from a single image using SHARP. |
|
||||
| `FisheyeToGaussian` | Reprojects a fisheye view to multiple pinhole angles, predicts splats, rotates and merges them. |
|
||||
| `RotateSplats` | Applies a 4×4 transform matrix to a splat cloud. |
|
||||
| `MergeSplats` | Concatenates two `GSPLAT` objects into one. |
|
||||
| `FuseSplats` | Fuses two splat clouds with weighted voxel merging (keep/discard/average/smart modes). |
|
||||
| `RenderSplat` | Renders a splat cloud from a camera pose into an image + mask. |
|
||||
| `VideoToFusedSplats` | Runs SHARP on video keyframes, scale-aligns to metric depth, filters dynamic pixels, and fuses all keyframes into one world-frame splat cloud. |
|
||||
| `SplatPolish` | Optimizes a world-frame splat cloud against posed video frames (L1 + D-SSIM) using gsplat's differentiable rasterizer. |
|
||||
| `MotionMaskFromDepth` | Detects dynamic pixels from a depth+pose sequence (1.0 = moving). |
|
||||
| `EstimateTracks` | Runs CoTracker3 on a video; returns tracks `[T,N,2]` (pixels) and visibility `[T,N]`. |
|
||||
| `TracksToTrajectories` | Unprojects 2D tracks with depth and camera poses into world-space 3D trajectories `[T,M,3]`. |
|
||||
| `SplitSplatsByMask` | Projects splat centers into a 2D mask and splits the cloud into inside/outside parts. |
|
||||
| `BuildSplats4D` | Builds a 4D splat scene: each canonical splat follows a kNN blend of track control-point motions. |
|
||||
| `RenderSplats4DFrame` | Evaluates the 4D scene at a single time value and renders it from a given camera. |
|
||||
| `RenderSplats4DVideo` | Interpolates the camera path, sweeps time from start to end, and renders each frame. |
|
||||
| `SaveSplats4D` | Saves a `GSPLAT4D` scene as an `.npz` archive (plus optional per-frame PLYs). |
|
||||
| `LoadSplats4D` | Loads a `GSPLAT4D` scene from an `.npz` archive. |
|
||||
| `VideoPoseEstimator` | VGGT-based per-frame camera poses `[T,4,4]`, depth maps, FOV and depth confidence from a video clip. |
|
||||
| `TrajectoryInvert` | Inverts each 4×4 pose (world-to-camera ↔ camera-to-world). |
|
||||
| `TrajectoryCompose` | Per-frame matrix product `A @ B`; a single 4×4 input broadcasts over the other. |
|
||||
| `DepthScaleAnchor` | Robustly aligns a depth map to a reference depth via disparity-domain scale(+shift). |
|
||||
| `SplatTrajectoryEnricher` | Expands a splat world along a trajectory: render, outpaint holes with Flux, lift with SHARP, scale-align, smart-stitch. |
|
||||
| `SphereSplatSeed` | Converts an equirectangular panorama into a Gaussian sphere seeding a 360° world. |
|
||||
|
||||
---
|
||||
|
||||
## Video → 4D World
|
||||
|
||||
Turn a monocular video into a navigable 4D (3D + time) Gaussian splat scene and re-render it from any novel camera trajectory. The reference workflow is **`workflows/video_to_4d_world.json`**; the stages are:
|
||||
|
||||
1. **Pose & depth (VGGT)** — `VideoPoseEstimator` estimates per-frame world-to-camera poses `[T,4,4]`, depth maps, FOV and depth confidence from the input frames. Since the depth maps are Z-depths, run `ZDepthToRayDepthNode` before any node that expects ray depth (see caveats below). `DepthEdgeFilter` can additionally mask out flying pixels at depth discontinuities.
|
||||
2. **Motion masking** — `MotionMaskFromDepth` warps depth between frames using the estimated poses and flags pixels whose residual is too large as dynamic (moving objects vs. static background).
|
||||
3. **Static splat fusion + polish** — `VideoToFusedSplats` runs SHARP on keyframes, keeps only static pixels (via the motion mask), scale-aligns each keyframe to metric depth, transforms splats into the world frame and fuses them incrementally. `SplatPolish` then fine-tunes the fused cloud photometrically against the posed video frames.
|
||||
4. **Tracked dynamic 4D Gaussians** — `EstimateTracks` (CoTracker3) tracks a dense point grid across the video; `TracksToTrajectories` lifts the tracks to world-space 3D using depth + poses; `SplitSplatsByMask` separates dynamic splats from the static background; `BuildSplats4D` binds the dynamic canonical splats to track control points via kNN blending, producing a `GSPLAT4D` scene.
|
||||
5. **Render a novel trajectory** — build any new camera path (e.g. `CameraInterpolationNode`, `TrajectoryCompose` to retarget relative to a source pose) and render with `RenderSplats4DVideo` (or single frames with `RenderSplats4DFrame`). Save/reload scenes with `SaveSplats4D` / `LoadSplats4D`.
|
||||
|
||||
**Static-camera fisheye variant** — for footage from a locked-off 180° fisheye camera, `workflows/fisheye_static_video_to_4d.json` skips pose estimation entirely (identity trajectory), uses the batched `FisheyeDepthEstimator` for per-frame radial depth and `FisheyeToGaussian` on frame 0 for the whole static world, then follows the same track → split → `BuildSplats4D` → render path (all 4D nodes accept the FISHEYE projection directly).
|
||||
|
||||
### Caveats
|
||||
|
||||
* **Z-depth vs ray depth**: depth estimators (including `VideoPoseEstimator`) output Z-depth; point-cloud and splat lifting nodes expect ray depth. Insert `ZDepthToRayDepthNode` where needed, or geometry will bow at wide FOVs.
|
||||
* **`SplatPolish` requires gsplat + CUDA**: without them it can fall back to the differentiable torch renderer at reduced resolution, which is extremely slow (minutes per 100 iterations).
|
||||
* **`EstimateTracks` downloads CoTracker3 via `torch.hub` on first use** — expect a one-time download and allow network access.
|
||||
* **`VideoPoseEstimator` downloads `facebook/VGGT-1B` (~5 GB)** on first use via `huggingface_hub`.
|
||||
|
||||
---
|
||||
|
||||
## Workflows
|
||||
|
||||
A set of JSON workflows illustrating typical use cases. Each workflow lives in `workflows/` and can be loaded directly in ComfyUI.
|
||||
A set of JSON workflows illustrating typical use cases. Once the pack is installed they appear in ComfyUI under **Workflow → Browse Templates** (with thumbnails); the files live in `workflows/` and can also be loaded directly. Each workflow contains an embedded **“About this workflow”** note in the canvas explaining its stages, what to set, and what it needs — and references the bundled example inputs that `install.py` copies into your ComfyUI `input/` folder, so they run as-is on a fresh install.
|
||||
|
||||
| Workflow | Description |
|
||||
| -------------------------------------- | -------------------------------------------------------------- |
|
||||
| **demo\_camera\_workflow\.json** | Masked reprojection demo: pinhole → fisheye/equirect |
|
||||
| **outpainting\_fisheye.json** | Text‐guided fisheye outpainting (built‐in inpaint node) |
|
||||
| **outpainting\_fisheye\_flux.json** | Flux‐based outpainting with clear reprojection scheme |
|
||||
| **Outpaint\_node\_test.json** | Test harness for the universal outpaint node |
|
||||
| **Outpaint\_fisheye180.json** | 180° fisheye outpainting via `OutpaintAnyProjection` |
|
||||
| **Fisheye\_depth\_workflow\.json** | Fisheye → metric depth → point cloud → PLY export |
|
||||
| **Pointcloud.json** | Metric‐depth‐anything v2 → point cloud → camera view synthesis |
|
||||
| **pointcloud\_inpaint.json** | Inpaint + backproject to 3D for dynamic camera motion videos |
|
||||
| **Pointcloud\_walker.json** | GUI‐based camera control via Open3D |
|
||||
| **sbs180\_workflow.json** | Generate stereo (side-by-side) wide-angle/fisheye/equirectangular stereo pairs from a high-res input |
|
||||
*Extras* below means dependencies beyond this pack and Depth-Anything V2 (which auto-downloads); `inpainting_flux` is installed automatically by `install.py`.
|
||||
|
||||
| Workflow | Description | Extras |
|
||||
| --- | --- | --- |
|
||||
| **demo\_camera\_workflow\.json** | Minimal demo: rotate the camera and reproject pinhole → equirectangular, with coverage mask | — |
|
||||
| **Outpaint\_node\_test.json** | One-patch smoke test of `OutpaintAnyProjection` | inpainting_flux |
|
||||
| **Outpaint\_fisheye180.json** | Pinhole 90° → full 180° fisheye via five chained `OutpaintAnyProjection` passes + composite/upscale | inpainting_flux |
|
||||
| **outpainting\_fisheye\_flux.json** | Manual version of the above: explicit Flux Inpainting + reprojection stages | inpainting_flux, RealESRGAN |
|
||||
| **fisheye\_to\_pointcloud.json** | Fisheye 180° → metric depth → point cloud (`.ply`/`.npy`) | — |
|
||||
| **PointCloud.json** | Single image → point cloud → cleaned novel-view render | Image-Filters (optional) |
|
||||
| **pointcloud\_walker.json** | Image → point cloud → camera fly-through WEBM | — |
|
||||
| **test\_pointcloud\_loading.json** | Reload a saved point cloud and orbit-render it | — |
|
||||
| **record\_trajectory.json** | Record a camera trajectory `.npy` for LoadTrajectory (two poses → SE(3) interpolation → SaveTrajectory) | — |
|
||||
| **pointcloud\_inpaint.json** | Enrich a cloud: Flux-inpaint disocclusions, lift them to 3D, merge, orbit render | inpainting_flux |
|
||||
| **PC\_enricher.json** | One-node version of the above: `PointcloudTrajectoryEnricher` along a saved trajectory | inpainting_flux |
|
||||
| **sbs180\_workflow.json** | Synthesize the second eye of a VR180 stereo pair from one fisheye view | inpainting_flux |
|
||||
| **video\_camera.json** | Re-shoot a video with a new camera move; WAN VACE regenerates disocclusions, Florence2 auto-captions | VHS, Florence2; WAN 2.1 VACE + Video-Depth-Anything models |
|
||||
| **wan\_vace\_ref\_to\_video.json** | Still fisheye image + recorded trajectory → WAN VACE camera-move video | WAN 2.1 VACE models; VHS (optional MP4 export) |
|
||||
| **video_to_4d_world\.json** | Video → 4D world: VGGT poses/depth → motion masking → fused static splats + polish → tracked dynamic 4D Gaussians → novel-trajectory render. | — |
|
||||
| **video_to_4d_walkable_world\.json** | Video → 4D WALKABLE world (test-friendly defaults): polished static splats enriched along a walk trajectory (`SplatTrajectoryEnricher`, Flux outpaint + SHARP) → 4D scene → walk-through render + `.ply`/`.npz` exports for free walking in external 3DGS viewers. | inpainting_flux |
|
||||
| **fisheye\_static\_video\_to\_4d.json** | Static-camera 180° fisheye video → 4D Gaussian scene → novel-path render. No pose estimation needed: identity trajectory, batched fisheye depth, frame-0 splats split into static world + dynamic canonical. | VHS |
|
||||
|
||||
Superseded reference graphs live in `workflows/legacy/` (kept out of the template browser): **outpainting\_fisheye.json** (SD-inpaint-checkpoint variant of the flux outpaint) and **Fisheye\_depth\_workflow\.json** (the multi-view depth fusion that `FisheyeDepthEstimator` now performs internally).
|
||||
|
||||
---
|
||||
|
||||
@@ -154,9 +261,9 @@ Basic reprojection pipeline: apply masks, rotate pinhole camera, outpaint fishey
|
||||
<img src="demo_images/Pinhole_camera_rotation.png" alt="Pinhole Rotation" width="45%" />
|
||||
</div>
|
||||
|
||||
### 2. `outpainting_fisheye.json`
|
||||
### 2. `legacy/outpainting_fisheye.json`
|
||||
|
||||
Simplest text‐guided fisheye outpainting built with the core inpaint node.
|
||||
Simplest text‐guided fisheye outpainting built with the core inpaint node (superseded by the Flux variant).
|
||||
|
||||
### 3. `outpainting_fisheye_flux.json`
|
||||
|
||||
@@ -172,9 +279,9 @@ Flux Inpainting ensures sharper results and explicit reprojection stages.
|
||||
|
||||
<img src="demo_images/Fisheye_outpainted_flux_dev.png" alt="Flux Dev" width="60%" />
|
||||
|
||||
### 5. `Fisheye_depth_workflow.json`
|
||||
### 5. `legacy/Fisheye_depth_workflow.json`
|
||||
|
||||
Convert fisheye images to metric depth and generate a PLY point cloud.
|
||||
Convert fisheye images to metric depth and generate a PLY point cloud — the manual multi-view graph that `FisheyeDepthEstimator` now performs in one node (see `fisheye_to_pointcloud.json`).
|
||||
|
||||
<img src="demo_images/Depthmap.png" alt="Fisheye Depth→PointCloud" width="60%" />
|
||||
|
||||
@@ -184,7 +291,7 @@ Convert fisheye images to metric depth and generate a PLY point cloud.
|
||||
|
||||
Quick test for the universal outpaint node in arbitrary views and camera movement
|
||||
|
||||
### 7. `Pointcloud.json`
|
||||
### 7. `PointCloud.json`
|
||||
|
||||
Depth→PointCloud pipeline with interactive camera movement and reprojection views.
|
||||
|
||||
@@ -204,15 +311,17 @@ Take a wide-angle (fisheye or equirectangular) high-resolution (e.g., 4096×4096
|
||||
|
||||
<img src="demo_images/equirect_stereo.gif" alt="Equirectangular Stereo Demo" width="80%" />
|
||||
|
||||
### 10. `Pointcloud_walker.json`
|
||||
### 10. `pointcloud_walker.json`
|
||||
|
||||
Interactive Open3D-based GUI for walking and setting camera trajectory inside pointcloud.
|
||||
Image → point cloud → camera fly-through rendered to WEBM (`CameraTrajectoryNode` + `CameraMotionNode`).
|
||||
|
||||
### 11. `wan-vace_ref_to_video.json`
|
||||
### 11. `video_camera.json`
|
||||
|
||||
Integrate the [wan2.1-vace] video generation model to inpaint empty or newly revealed regions during camera movement or view synthesis. This workflow demonstrates how to use the camera-comfyUI nodes to generate camera trajectories and masks, then fill missing areas with the video inpainting model for smooth, high-quality results.
|
||||
This workflow demonstrates camera trajectory movement using the `wan-vace` video inpainting model. It generates smooth camera movements along a trajectory while filling missing regions with high-quality inpainting.
|
||||
|
||||
<img src="demo_images/wan-vace-camera.gif" alt="wan2.1-vace Camera Inpainting Demo" width="80%" />
|
||||
<div style="display:flex; gap:10px;">
|
||||
<img src="demo_images/camera_movement.gif" alt="Camera Movement Demo" width="80%" />
|
||||
</div>
|
||||
|
||||
---
|
||||
|
||||
@@ -256,9 +365,12 @@ Contributions welcome! Please open issues or PRs to add features, improve docs,
|
||||
* [x] Add processing to pointcloud or depthmap to remove outlier and lonely points at depth borders.
|
||||
* [x] Use built-in comfyUI mask type an image.
|
||||
* [x] Unite nodes into groups to simplify workflows.
|
||||
* [ ] Create a single workflow for view synthesis.
|
||||
* [x] Create a single workflow for view synthesis (`video_to_4d_world.json`).
|
||||
* [x] Implement easier and more flexible camera control - more complex camera movements with more than 2 points.
|
||||
* [x] Add more examples and documentation for each node.
|
||||
* [x] Add pointcloud union
|
||||
* [x] Fix imports for renamed folders (e.g., inpainting_flux)
|
||||
* [x] Integrate camera movement pipeline with video models (e.g., wan2.1) for smooth, high-quality inpainting along camera trajectories.
|
||||
* [ ] Compressed export format for 4D scenes (current `.npz` stores raw tensors).
|
||||
* [ ] SAM2-based refinement of motion masks (current masks come from depth-warp residuals only).
|
||||
* [ ] Fisheye/equirectangular rendering through gsplat (e.g., via cubemap render + reprojection); the fast CUDA path is currently pinhole-only.
|
||||
|
||||
@@ -4,6 +4,27 @@ from .metric_depth_nodes import NODE_CLASS_MAPPINGS as NCM3
|
||||
from .flux_fisheye_filling_nodes import NODE_CLASS_MAPPINGS as NCM4
|
||||
from .complex_nodes import NODE_CLASS_MAPPINGS as NCM5
|
||||
from .video_nodes import NODE_CLASS_MAPPINGS as NCM6
|
||||
NODE_CLASS_MAPPINGS = {**NCM1, **NCM2, **NCM3, **NCM4, **NCM5, **NCM6}
|
||||
from .GS_nodes import NODE_CLASS_MAPPINGS as NCM7
|
||||
|
||||
__all__ = ["NODE_CLASS_MAPPINGS"]
|
||||
# Optional node packs: a missing/broken optional dependency must never kill the
|
||||
# whole extension (mirrors how video_nodes degrades when video_depth_anything
|
||||
# is unavailable).
|
||||
try:
|
||||
from .GS4D_nodes import NODE_CLASS_MAPPINGS as NCM8
|
||||
except Exception as _exc:
|
||||
print(f"[camera-comfyUI] Warning: GS4D_nodes could not be loaded, 4D splat nodes disabled: {_exc}")
|
||||
NCM8 = {}
|
||||
try:
|
||||
from .pose_nodes import NODE_CLASS_MAPPINGS as NCM9
|
||||
except Exception as _exc:
|
||||
print(f"[camera-comfyUI] Warning: pose_nodes could not be loaded, pose estimation nodes disabled: {_exc}")
|
||||
NCM9 = {}
|
||||
try:
|
||||
from .world_nodes import NODE_CLASS_MAPPINGS as NCM10
|
||||
except Exception as _exc:
|
||||
print(f"[camera-comfyUI] Warning: world_nodes could not be loaded, world-building nodes disabled: {_exc}")
|
||||
NCM10 = {}
|
||||
|
||||
NODE_CLASS_MAPPINGS = {**NCM1, **NCM2, **NCM3, **NCM4, **NCM5, **NCM6, **NCM7, **NCM8, **NCM9, **NCM10}
|
||||
|
||||
__all__ = ["NODE_CLASS_MAPPINGS"]
|
||||
|
||||
|
After Width: | Height: | Size: 12 MiB |
@@ -0,0 +1,127 @@
|
||||
# LingBot-World 2.0 → 4D video: analysis & integration report
|
||||
|
||||
*Research date: 2026-07-13. LingBot-World 2.0 was released 2026-07-09, four days before this report.*
|
||||
|
||||
## TL;DR
|
||||
|
||||
**LingBot-World 2.0 is not a 3D/4D model — it is a camera-pose- and action-conditioned autoregressive video generator.** It outputs only pixels and maintains no explicit geometry. But it has exactly the property that makes a video-generation model useful for 4D reconstruction: **you command the camera trajectory (poses + intrinsics) of every generated frame**, so every output video is a *posed* video. That turns it into a controllable multi-view video factory whose output can be lifted into 4D Gaussian splats by the existing `video_to_4d_world.json` pipeline in this repo — with the pose-estimation step optionally replaced by the commanded poses.
|
||||
|
||||
Feasibility verdicts:
|
||||
|
||||
| Question | Verdict |
|
||||
| --- | --- |
|
||||
| 4D video from a 3D scene (splat/mesh) | **Yes, indirectly** — render the 3D scene to a seed image, then LingBot animates + explores it. 3D enters only as a rendered start frame; there is no native 3D conditioning. |
|
||||
| 4D Gaussian-splat video from its output | **Feasible and first-party-endorsed** — the LingBot-World paper itself demonstrates reconstructing its generated videos into point clouds with VGGT-class models, the same VGGT this repo already uses. |
|
||||
| Drop-in ComfyUI use today | **Not yet** — 14B Wan2.2-based weights, no quantized release for v2, no wrapper support yet ([kijai/WanVideoWrapper#1920](https://github.com/kijai/ComfyUI-WanVideoWrapper/issues/1920), [Comfy-Org/ComfyUI#12154](https://github.com/Comfy-Org/ComfyUI/issues/12154)); reference inference is 8×GPU `torchrun`. |
|
||||
| Commercial use | **v2: no** (CC BY-NC-SA 4.0). **v1: yes** (Apache 2.0). This alone may decide which version to build on. |
|
||||
|
||||
---
|
||||
|
||||
## 1. What LingBot-World 2.0 actually is
|
||||
|
||||
**Repos & papers**
|
||||
- v2 (current): [Robbyant/lingbot-world-v2](https://github.com/Robbyant/lingbot-world-v2) — "Infinite Worlds with Versatile Interactions", tech report [arXiv:2607.07534](https://arxiv.org/abs/2607.07534), weights [robbyant/lingbot-world-v2-14b-causal-fast](https://huggingface.co/robbyant/lingbot-world-v2-14b-causal-fast). Released 2026-07-09 by Robbyant (embodied-AI subsidiary of Ant Group).
|
||||
- v1 (deprecated but still useful): [Robbyant/lingbot-world](https://github.com/robbyant/lingbot-world) — "Advancing Open-source World Models", [arXiv:2601.20540](https://arxiv.org/abs/2601.20540), weights `robbyant/lingbot-world-base-cam` / `-base-act` / `-fast`. Released 2026-01-29.
|
||||
|
||||
**Architecture (verified against code + paper)**
|
||||
- Built on **Wan2.2 i2v-A14B**: a two-expert MoE video diffusion model, ~28B total parameters with **14B active** per denoising step (high-noise expert for global structure, low-noise for detail). Ships the Wan2.1 VAE and umT5-XXL text encoder.
|
||||
- v2 converts it to **causal, chunk-by-chunk autoregressive generation**: latents are generated `chunk_size` latent frames at a time against a **KV cache** with **sink tokens** and a **local attention window** (`run_fast.sh` uses `--local_attn_size 18 --sink_size 6`). A **MoBA mask** ("Mixture of Bidirectional and Autoregressive Attention Mask") mixes bidirectional attention into teacher forcing to stop the long-horizon quality collapse that plagues autoregressive video. Result: the paper demonstrates an **uninterrupted hour-long session with no perceptible quality decay**.
|
||||
- Two inference modes: `causal_fast` (distilled few-step; drives **720p @ 60 fps** in their real-time deployment) and `causal_pretrain` (40-step CFG; checkpoint still marked TODO). A single-GPU **1.3B variant is described in the paper but not released**.
|
||||
|
||||
**Conditioning inputs — the part that matters for 4D** (from `wan/image2video.py` + `wan/utils/cam_utils.py`)
|
||||
- **Seed image** (`--image`) + **text prompt**: the world is initialized from one image and a background description. This is the *only* way content enters — no 3D input of any kind.
|
||||
- **Camera trajectory**: `poses.npy` `[T,4,4]` **camera-to-world, OpenCV convention** + `intrinsics.npy` `[T,4]` = `[fx,fy,cx,cy]`. Converted to per-pixel **Plücker ray embeddings** (`get_plucker_embeddings`), folded into the latent grid and injected per-chunk into the DiT (AdaLN per the tech report). Relative poses are translation-normalized (`compute_relative_poses`), and `interpolate_camera_poses` (SLERP) is provided.
|
||||
- **Keyboard actions**: `wasd_action.npy` (movement) / `ijkl_action.npy` (view) as multi-hot vectors concatenated onto the Plücker conditioning. v2 adds character actions (attack, archery, spell-cast, shoot, jump, glide) and **chunk-wise text events** (weather, entity spawning, time-of-day), plus a VLM-driven "pilot/director" agentic harness.
|
||||
- v1 README explicitly recommends **[NVIDIA ViPE](https://github.com/nv-tlabs/vipe)** to extract `poses.npy`/`intrinsics.npy` from an *existing real video* — i.e., the official video→control-signal bridge.
|
||||
|
||||
**Inference & hardware**
|
||||
```bash
|
||||
torchrun --nproc_per_node=8 generate.py --task i2v-A14B --size 480*832 \
|
||||
--frame_num 361 --ckpt_dir lingbot-world-v2-14b-causal-fast \
|
||||
--image examples/03/image.jpg --action_path examples/03 \
|
||||
--infer_mode causal_fast --dit_fsdp --t5_fsdp --ulysses_size 8 \
|
||||
--local_attn_size 18 --sink_size 6
|
||||
```
|
||||
- Reference: 8×GPU (FSDP + Ulysses sequence parallel), 480×832, 361 frames (`frame_num` must be 4n+1). Single-GPU runs auto-enable `--offload_model` (T5/DiT swapped to CPU between stages) — expect 80GB-class VRAM for comfortable 14B bf16 inference; there is **no quantized v2 release yet**. v1 has a community **4-bit quant** and `--t5_cpu`, and supports up to 961 frames (~1 min @ 16 fps).
|
||||
- Requirements: `torch >= 2.4.0`, `flash_attn`.
|
||||
|
||||
**License** — v2 code *and* weights are **CC BY-NC-SA 4.0 (non-commercial, share-alike)**; v1 is **Apache 2.0**. Anything commercial built on v2 outputs is off the table; v1 remains the commercially safe option at lower quality/horizon.
|
||||
|
||||
---
|
||||
|
||||
## 2. Can it turn 3D into 4D video?
|
||||
|
||||
**Yes, with the 3D scene entering as a rendered image, not as geometry.** The paper is explicit that the world "is initialized from an initial image and its background description" — there is no splat/mesh/point-cloud conditioning path, and the model "operates without an explicit notion of geometry."
|
||||
|
||||
The working recipe, using nodes already in this repo:
|
||||
|
||||
1. **Render a seed view** of your static 3D asset: `LoadPlySplat` → `RenderSplat` (or a mesh render) at 832×480+, from a pose with good scene coverage.
|
||||
2. **Author the camera trajectory you want** in the splat's own coordinate frame (`CameraInterpolationNode` / `CameraTrajectoryNode`), convert to camera-to-world OpenCV `poses.npy` + `intrinsics.npy`.
|
||||
3. **Feed image + poses + actions/text-events to LingBot-World.** The model animates the scene (wind, characters, weather, spawned entities via text events) while following your camera — i.e., it *invents plausible dynamics* for your static 3D scene. This is "3D → 4D video" in the sense of *generating* the time dimension, not simulating it: physics is learned and imperfect, and the output will drift from your 3D asset's exact geometry the further the camera goes from the seed view.
|
||||
4. **Optionally lift the result back to 4D splats** (section 3) so the animated version of your scene becomes re-renderable from any camera.
|
||||
|
||||
Caveat on fidelity: only the seed frame is constrained by your 3D input. Occluded/unseen regions are hallucinated. For higher fidelity to the source scene you can seed successive generations from renders at multiple poses and stitch — the same strategy `SplatTrajectoryEnricher` already uses with Flux outpainting, but with LingBot providing temporally coherent *video* instead of stills.
|
||||
|
||||
---
|
||||
|
||||
## 3. Feasibility: 4D Gaussian-splat video from LingBot output
|
||||
|
||||
**This is the strongest part of the story.** Three findings, all verified against primary sources:
|
||||
|
||||
1. **Posed video for free.** Because generation is conditioned on `poses.npy`/`intrinsics.npy`, every generated frame comes with a commanded camera. A monocular real video gives you poses only after VGGT/COLMAP estimation; LingBot gives you the trajectory you asked for. (Treat commanded poses as *approximate* — the model follows them but is not geometrically exact; see limitations.)
|
||||
2. **First-party evidence that reconstruction works.** The LingBot-World paper itself demonstrates: *"by leveraging large-scale 3D reconstruction foundation models [lin2025depth, wang2025vggt], we can further convert the generated video sequences into high-quality scene point clouds"*, with point clouds showing *"strong spatial coherence across frames"* (Fig. 16, [arXiv:2601.20540](https://arxiv.org/html/2601.20540v1)). That is literally VGGT — the model behind this repo's `VideoPoseEstimator` — applied to LingBot output by its own authors.
|
||||
3. **Long-horizon consistency is the v2 headline.** Landmarks stay structurally intact after being out of view for up to ~60 s (v1) and v2 extends coherent generation to hour scale with no perceptible decay. Long consistent orbits are exactly what splat optimization needs.
|
||||
|
||||
**How it maps onto known video-to-4D paradigms:**
|
||||
- **CAT4D-style** ([arXiv:2411.18613](https://arxiv.org/abs/2411.18613)): camera/time-disentangled video diffusion → deformable 3DGS optimization. LingBot is not time-disentangled (you cannot freeze time and move the camera — camera and time advance together in one causal stream), so you *cannot* get true simultaneous multi-view of a dynamic instant from a single run.
|
||||
- **Monocular 4D lifting** (this repo's pipeline): works on any single posed video — LingBot output qualifies directly and improves on real footage by letting you *choose* a camera path that orbits/parallaxes around the action, which is the single biggest quality lever for monocular 4D reconstruction.
|
||||
- **Multi-run multi-view**: re-running with the same seed image but different trajectories gives multiple views of the *same static scene* but **different sampled dynamics** (different seeds/action outcomes per run) — usable for static splat fusion, **not** for dynamic 4D supervision. Keep dynamics within one continuous run.
|
||||
|
||||
**Bottom line:** treat LingBot-World as a *trajectory-controllable monocular video source* feeding the existing 4D pipeline; don't expect synchronized multi-view rigs out of it.
|
||||
|
||||
---
|
||||
|
||||
## 4. Concrete pipeline: video → 4D video / 4D splats
|
||||
|
||||
### Path A — real video in, 4D world out, LingBot as the world extender
|
||||
|
||||
Your existing `video_to_4d_world.json` already handles real-video → 4D. LingBot adds value where that pipeline is weakest: viewpoints the source video never saw.
|
||||
|
||||
1. **Base 4D scene from the real video** (existing flow): `VideoPoseEstimator` (VGGT poses/depth) → `ZDepthToRayDepthNode` → `MotionMaskFromDepth` → `VideoToFusedSplats` + `SplatPolish` (static) → `EstimateTracks`/`TracksToTrajectories`/`SplitSplatsByMask`/`BuildSplats4D` (dynamic) → `GSPLAT4D`.
|
||||
2. **Extract control signals from the same video** with ViPE (officially recommended) or reuse the VGGT poses: `VideoPoseEstimator` outputs world-to-camera `[T,4,4]` → `TrajectoryInvert` → camera-to-world OpenCV → export `poses.npy` + `intrinsics.npy` (VGGT's FOV output gives `fx,fy`; `cx,cy` = image center). *(Small new node needed: `TrajectoryToNpyExport` — trivial, ~20 lines.)*
|
||||
3. **Continue the world where the video ends**: last real frame = LingBot seed image; author an exploration trajectory (orbit, dolly, walk) continuing from the last real pose; generate 361+ frames.
|
||||
4. **Lift the generated segment** through the same stage-1 flow and **fuse into the base scene**: `FuseSplats`/`MergeSplats` for statics (scale-anchor with `DepthScaleAnchor` against the base scene's depth), separate `BuildSplats4D` time range for new dynamics. Result: a 4D world larger than the source footage.
|
||||
|
||||
### Path B — single image or 3D scene in, 4D splat video out
|
||||
|
||||
1. **Seed**: any image, or a render of an existing splat (`RenderSplat`) / mesh.
|
||||
2. **Trajectory design**: slow orbit or arc around the subject + gentle forward motion — maximize parallax, avoid pure rotation (no baseline → no geometry). Keep FOV fixed; write `poses.npy`/`intrinsics.npy` (c2w, OpenCV; translations get normalized internally, so keep the trajectory scale moderate and re-anchor metric scale later with `DepthScaleAnchor`).
|
||||
3. **Generate** with `causal_fast`, 480×832, 361 frames; drive dynamics with keyboard/character actions and chunk-wise text events ("a horse gallops through", "rain starts").
|
||||
4. **Reconstruct** — two pose options:
|
||||
- *Trust-but-verify (recommended)*: run `VideoPoseEstimator` on the generated frames anyway; compare with commanded poses (`TrajectoryCompose` of one with `TrajectoryInvert` of the other should be ≈ identity); use VGGT's poses for reconstruction, commanded poses as sanity check. This absorbs the model's camera-following error.
|
||||
- *Fast path*: use commanded poses directly, skip VGGT pose estimation, still run its depth head (or `VideoMetricDepthEstimate`) for the depth maps the lifting nodes need.
|
||||
5. **Lift to 4D**: identical to the existing workflow — motion mask → static fusion (`VideoToFusedSplats` + `SplatPolish`) → tracks (`EstimateTracks` is CoTracker3, works fine on generated footage) → `BuildSplats4D` → `RenderSplats4DVideo` along any novel camera path → `SaveSplats4D`.
|
||||
|
||||
### Integration notes for camera-comfyUI
|
||||
|
||||
- **Coordinate conventions align well**: LingBot uses OpenCV c2w + `[fx,fy,cx,cy]`, this repo's `TRAJECTORY` is 4×4 matrices with `TrajectoryInvert`/`TrajectoryCompose` already available. Needed glue: (a) `TrajectoryToNpyExport` / `NpyToTrajectory` nodes, (b) optionally a `LingBotGenerate` node wrapping `generate.py` via subprocess for remote/8-GPU boxes — running 14B in-process inside ComfyUI is not realistic today.
|
||||
- **ComfyUI-native inference isn't there yet**: WanVideoWrapper/ComfyUI support for LingBot checkpoints is an open request blocked on VRAM/quantization ([#1920](https://github.com/kijai/ComfyUI-WanVideoWrapper/issues/1920), [#12154](https://github.com/Comfy-Org/ComfyUI/issues/12154)). Because it's Wan2.2-architecture, wrapper support and GGUF/FP8 quants are likely to appear quickly; the causal KV-cache/sink/MoBA inference loop is custom, so a naive Wan2.2 loader won't reproduce long-horizon behavior.
|
||||
- **Pragmatic hardware ladder**: (1) today, single-image experiments on v1 `base-cam` 4-bit quant (Apache 2.0, 480p, camera-pose conditioned — same poses.npy interface) on a 24 GB GPU; (2) v2 14B on a rented 8×A100/H100 node or single 80 GB GPU with offload; (3) wait for the announced 1.3B v2 release for true single-GPU local use.
|
||||
|
||||
### Known limitations
|
||||
|
||||
- **No geometry inside the model** — all 3D/4D structure comes from post-hoc reconstruction; physics is "imperfect" by the authors' own admission.
|
||||
- **Camera-following error**: commanded poses ≠ achieved poses exactly (Plücker conditioning is a soft constraint; translations are normalized, so absolute scale is undefined) — always re-anchor scale and consider re-estimating poses.
|
||||
- **Dynamics are not repeatable across runs** — multi-view supervision of a dynamic instant is impossible; design single continuous runs whose camera moves *around* the action.
|
||||
- **480×832 native offline resolution** (720p is the real-time streaming mode) — plan on splat-space upscaling or `SplatPolish` against upscaled frames.
|
||||
- **Generated-content artifacts** (texture shimmer, occasional object morphing) become floaters/ghosts in splat space — the existing `MotionMaskFromDepth` + `DepthEdgeFilter` + `PointCloudCleaner` stack mitigates this, and track-validity filtering in `TracksToTrajectories` matters more than with real footage.
|
||||
- **License**: v2 is CC BY-NC-SA 4.0 — non-commercial only, share-alike. Use v1 (Apache 2.0) for anything with commercial intent.
|
||||
|
||||
---
|
||||
|
||||
## Sources
|
||||
|
||||
Primary: [lingbot-world-v2 repo](https://github.com/Robbyant/lingbot-world-v2) · [v2 tech report arXiv:2607.07534](https://arxiv.org/abs/2607.07534) · [v2 weights (HF)](https://huggingface.co/robbyant/lingbot-world-v2-14b-causal-fast) · [lingbot-world v1 repo](https://github.com/robbyant/lingbot-world) · [v1 paper arXiv:2601.20540](https://arxiv.org/abs/2601.20540) · [v1 cam weights (HF)](https://huggingface.co/robbyant/lingbot-world-base-cam) · code files `generate.py`, `wan/image2video.py`, `wan/utils/cam_utils.py`, `run_fast.sh` (read directly).
|
||||
Secondary: [Robbyant press release (2026-07-09)](https://www.businesswire.com/news/home/20260708757367/en/Robbyant-Unveils-LingBot-World-2.0-Pioneering-Hour-Long-Real-Time-Generation-in-World-Models) · [v1 release (2026-01-28)](https://www.businesswire.com/news/home/20260128459962/en/Robbyant-Open-Sources-LingBot-World-a-World-Model-for-Millisecond-Level-Real-Time-Interaction) · [CAT4D arXiv:2411.18613](https://arxiv.org/abs/2411.18613) · [ViPE](https://github.com/nv-tlabs/vipe) · ComfyUI support threads [WanVideoWrapper#1920](https://github.com/kijai/ComfyUI-WanVideoWrapper/issues/1920), [ComfyUI#12154](https://github.com/Comfy-Org/ComfyUI/issues/12154).
|
||||
|
||||
*Method note: claims were gathered by a fan-out research pass (18 sources, 90 raw claims, 25 adversarially verified: 14 confirmed 3-0, 3 refuted, 8 verification-errored) plus direct reading of both repos' inference code and both arXiv papers. The two load-bearing claims whose automated verification errored (v1's video→point-cloud demonstration; the unreleased 1.3B variant) were re-verified manually against the arXiv HTML.*
|
||||
@@ -0,0 +1,96 @@
|
||||
# Workflow review — July 2026
|
||||
|
||||
Scope: all 17 `workflows/*.json`. Tooling used (kept in `notebooks/`):
|
||||
|
||||
* `validate_workflows.py` — checks every stored `widgets_values` array against
|
||||
the *current* `INPUT_TYPES` of the node it targets (count + combo values).
|
||||
Run it whenever a node's inputs change.
|
||||
* `rework_workflows_2026_07.py` — the one-shot migration that produced the
|
||||
current state of the files (documented below, idempotent-ish).
|
||||
|
||||
## What was wrong (now fixed)
|
||||
|
||||
ComfyUI applies `widgets_values` **positionally**. When a node gains/loses/
|
||||
reorders widgets, old workflows load with silently shifted values — no error,
|
||||
just wrong settings. The validator found 28 such cases across 9 files:
|
||||
|
||||
| Node | Old → new widgets | Files affected | Migration applied |
|
||||
| --- | --- | --- | --- |
|
||||
| `DepthEstimatorNode` | 2 → 3 (`median_blur_kernel` added) | 5 files, 12 nodes | appended default `1` |
|
||||
| `FisheyeDepthEstimator` | 6 → 8 (`mode`, `median_blur_kernel` added) | 2 files | inserted `SOFTMERGE` (radius was already configured), appended `1` |
|
||||
| `PointCloudCleaner` | 2 → 4 (screen-space `width`/`height` added; units changed) | `PointCloud.json` | reset to defaults `[1024, 1024, 1.0, 3]` — old world-unit values were meaningless in the new schema (**retune on GPU box**) |
|
||||
| `CameraMotionNode` | 6 → 9 (`widen_mask`, `invert_mask`, `points_to_mask`) | 3 files | appended defaults `[0, false, false]` |
|
||||
| `CameraInterpolationNode` | 0 → 1 (`num_steps`) | 3 files | set `2` (keyframes only — `CameraMotionNode`/`VideoCameraMotionSequence` interpolate frames themselves) |
|
||||
| `PointcloudTrajectoryEnricher` | 20 → 16 (render/reproject-back options internalized) | `PC_enricher.json` | first 13 kept 1:1, new tail set to defaults |
|
||||
|
||||
Beyond widget drift:
|
||||
|
||||
* **`outpainting_fisheye.json` was structurally broken**: its seven
|
||||
`ReprojectImage` nodes stored patch rotations (±42°) in a pre-2025 widget
|
||||
slot that now lands on the `inverse` boolean (42 → truthy). Repaired by
|
||||
adding two `TransformToMatrix` nodes (±42°) wired to `transform_matrix`
|
||||
inputs and setting proper `inverse` flags — mirroring the structure of
|
||||
`outpainting_fisheye_flux.json`.
|
||||
* **`outpainting_fisheye_flux.json`** stored `45` in one `inverse` slot
|
||||
(truthy-by-accident); normalized all seven to real booleans.
|
||||
* **`wan_vace_ref_to_video.json`** pointed the UNETLoader at
|
||||
`wan2,1_vace14B_fp16.safetensors` (comma typo + missing underscore) — no such
|
||||
file can exist; fixed to `wan2.1_vace_14B_fp16.safetensors` (matches
|
||||
`install.sh`'s download name).
|
||||
* `Pointcloud walker.json` / `Test pointcloud_loading.json` renamed to
|
||||
`pointcloud_walker.json` / `test_pointcloud_loading.json` (spaces break
|
||||
shell ergonomics and URL linking).
|
||||
* README workflow table was stale (claimed `pointcloud_walker` was an "Open3D
|
||||
GUI", missed 5 workflows); rewritten with per-workflow extras columns.
|
||||
|
||||
Every workflow now carries an embedded **“About this workflow”** MarkdownNote
|
||||
(purpose, stages, what to set, required packs/models), meaningful group boxes,
|
||||
and titles on the nodes users are expected to edit.
|
||||
|
||||
## Improvements — applied 2026-07-16
|
||||
|
||||
Implemented by `notebooks/apply_improvements_2026_07.py` plus repo changes:
|
||||
|
||||
1. **Runnable defaults** ✅ — `example_inputs/` ships
|
||||
`camera_example_pinhole.jpg`, `camera_example_fisheye.jpg` and
|
||||
`ComfyUITrajectory_00001.npy`; `install.py` copies them into ComfyUI's
|
||||
`input/` dir (no overwrite), and every LoadImage/LoadTrajectory default
|
||||
points at them. Exception: `video_camera.json` still needs a user video
|
||||
(none bundled — a clip would bloat the archive).
|
||||
2. **Template browser** ✅ — `workflows/` is already an accepted template
|
||||
directory name (per docs.comfy.org, alongside `example_workflows`), so no
|
||||
rename was needed; added the missing same-name `.jpg` thumbnails (10
|
||||
workflows, generated 512 px from `demo_images/`) and removed the stray
|
||||
`workflows/__init__.py`. Legacy graphs moved to `workflows/legacy/`, which
|
||||
keeps them out of the browser.
|
||||
3. **`video_camera.json` dep trim** ✅ — removed the dead-end `BlurMaskFast`
|
||||
(blur radius was 0/0 — a no-op even if wired), the `easy mathInt` pad
|
||||
computation and the then-dangling `VHS_VideoInfo`; pad top/bottom now use
|
||||
the node's stored values (set to `(W−H)/2`, note explains); swapped
|
||||
KJNodes' `GetImageRangeFromBatch` for the built-in `ImageFromBatch`.
|
||||
Remaining pack deps: VideoHelperSuite + Florence2 (was five packs).
|
||||
4. **Fisheye-outpaint consolidation** ✅ — SD-checkpoint variant moved to
|
||||
`workflows/legacy/outpainting_fisheye.json`.
|
||||
5. **Depth consolidation** ✅ — `workflows/legacy/Fisheye_depth_workflow.json`.
|
||||
6. **Trajectory recorder** ✅ — new `workflows/record_trajectory.json`
|
||||
(TransformToMatrix ×2 → CameraInterpolationNode → SaveTrajectory), and the
|
||||
bundled example trajectory covers the zero-setup path.
|
||||
7. **CI guard** ✅ — `.github/workflows/validate.yml` runs the workflow
|
||||
validator, installer-logic tests and the 4D smoke suite on CPU torch for
|
||||
every PR and push to main.
|
||||
8. **Schema versioning** ✅ — every workflow now carries
|
||||
`extra.camera_comfyui_rev = 1`; bump on the next migration.
|
||||
9. Bonus: a bidirectional link-integrity sweep found and pruned 8 stale link
|
||||
references (pre-existing) plus one dangling `mask` input in
|
||||
`outpainting_fisheye_flux.json`.
|
||||
|
||||
## Still open
|
||||
|
||||
* **Retune migrated defaults on the GPU box.** `PointCloudCleaner` (in
|
||||
`PointCloud.json`) and the voxel-merge tail of `PointcloudTrajectoryEnricher`
|
||||
(in `PC_enricher.json`) were reset to schema defaults; verify visual quality
|
||||
and bake in good values.
|
||||
* **Load-and-queue pass in real ComfyUI.** All checks here are static; open
|
||||
each template once on the GPU box to confirm layout and execution.
|
||||
* **Bundle a small example video** for `video_camera.json` if archive size
|
||||
allows (or document a public sample clip URL in its note).
|
||||
|
After Width: | Height: | Size: 1.3 MiB |
|
After Width: | Height: | Size: 196 KiB |
@@ -4,16 +4,52 @@ import numpy as np
|
||||
# reprojection helpers
|
||||
from .reprojection_nodes import Projection, ReprojectImage, TransformToMatrix
|
||||
|
||||
# Try importing FluxInpainting and capture any ImportError
|
||||
import importlib
|
||||
import sys, os, logging
|
||||
here = os.path.dirname(os.path.dirname(os.path.dirname(__file__)))
|
||||
if here not in sys.path:
|
||||
sys.path.append(here)
|
||||
|
||||
# Folder names the ComfyUI-Flux-Inpainting pack may live under in custom_nodes/
|
||||
# (install.py clones it as "inpainting_flux"; keep both lists in sync).
|
||||
_FLUX_PACK_CANDIDATES = (
|
||||
"inpainting_flux",
|
||||
"ComfyUI-Flux-Inpainting",
|
||||
"ComfyUI-Flux-Inpainting-main",
|
||||
"comfyui-flux-inpainting",
|
||||
)
|
||||
|
||||
|
||||
def _import_flux_inpainting():
|
||||
"""Import FluxNF4Inpainting from the flux inpainting pack regardless of the
|
||||
folder name it was installed under."""
|
||||
custom_nodes_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
comfy_root = os.path.dirname(custom_nodes_dir)
|
||||
if comfy_root not in sys.path:
|
||||
sys.path.append(comfy_root)
|
||||
|
||||
last_error = None
|
||||
for name in _FLUX_PACK_CANDIDATES:
|
||||
if not os.path.isfile(os.path.join(custom_nodes_dir, name, "nodes.py")):
|
||||
continue
|
||||
try:
|
||||
module = importlib.import_module(f"custom_nodes.{name}.nodes")
|
||||
return module.FluxNF4Inpainting
|
||||
except Exception as exc:
|
||||
last_error = exc
|
||||
if last_error is None:
|
||||
last_error = ModuleNotFoundError(
|
||||
"No flux inpainting pack found in custom_nodes (looked for "
|
||||
f"{', '.join(_FLUX_PACK_CANDIDATES)}). It is installed automatically "
|
||||
"by this pack's install.py (run by ComfyUI-Manager); to set it up "
|
||||
f"manually, clone {'https://github.com/rubi-du/ComfyUI-Flux-Inpainting'} "
|
||||
"into custom_nodes/inpainting_flux."
|
||||
)
|
||||
raise last_error
|
||||
|
||||
|
||||
# Try importing FluxInpainting and capture any error
|
||||
try:
|
||||
from custom_nodes.inpainting_flux.nodes import FluxNF4Inpainting as FluxInpainting
|
||||
FluxInpainting = _import_flux_inpainting()
|
||||
_flux_import_error = None
|
||||
except ImportError as e:
|
||||
except Exception as e:
|
||||
FluxInpainting = None
|
||||
_flux_import_error = e
|
||||
logging.error(f"[OutpaintAnyProjection] could not import FluxNF4Inpainting: {e}")
|
||||
@@ -71,7 +107,8 @@ class OutpaintAnyProjection:
|
||||
if _flux_import_error is not None:
|
||||
raise RuntimeError(
|
||||
f"FluxNF4Inpainting is not available: {_flux_import_error}\n"
|
||||
"Please install or fix your inpainting_flux package."
|
||||
"Run this pack's install.py (ComfyUI-Manager does this automatically) "
|
||||
"or install/fix custom_nodes/inpainting_flux."
|
||||
)
|
||||
|
||||
def normalize_mask(m: torch.Tensor):
|
||||
|
||||
@@ -0,0 +1,215 @@
|
||||
"""Post-install setup for camera-comfyUI.
|
||||
|
||||
ComfyUI-Manager runs this script automatically after installing the node pack
|
||||
(both git and registry installs), right after `pip install -r requirements.txt`.
|
||||
Manual users can run it themselves: `python install.py` from this directory.
|
||||
|
||||
It makes the optional heavy dependencies work out of the box:
|
||||
* vggt — pip-installed from GitHub (not on PyPI); needed by VideoPoseEstimator.
|
||||
* SHARP — the submodules/ml-sharpt checkout; needed by ImageToSplat & co.
|
||||
* gsplat — SHARP's rasterizer backend (JIT-compiles CUDA kernels on first use).
|
||||
* inpainting_flux — the ComfyUI-Flux-Inpainting node pack, cloned as a sibling
|
||||
custom node; needed by OutpaintAnyProjection / SplatTrajectoryEnricher.
|
||||
|
||||
Every step is idempotent and non-fatal: the node pack degrades gracefully at
|
||||
runtime (see __init__.py), so a failed optional step only disables the nodes
|
||||
that need it. The script therefore always exits 0 and reports what it skipped.
|
||||
"""
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
NODE_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
LOG_PREFIX = "[camera-comfyUI install]"
|
||||
|
||||
VGGT_GIT_URL = "https://github.com/facebookresearch/vggt.git"
|
||||
SHARP_GIT_URL = "https://github.com/apple/ml-sharp"
|
||||
FLUX_PACK_GIT_URL = "https://github.com/rubi-du/ComfyUI-Flux-Inpainting.git"
|
||||
# Folder names under custom_nodes/ that count as "the flux inpainting pack is
|
||||
# already installed" (must stay in sync with flux_fisheye_filling_nodes.py).
|
||||
FLUX_PACK_CANDIDATES = (
|
||||
"inpainting_flux",
|
||||
"ComfyUI-Flux-Inpainting",
|
||||
"ComfyUI-Flux-Inpainting-main",
|
||||
"comfyui-flux-inpainting",
|
||||
)
|
||||
|
||||
|
||||
def _log(message: str) -> None:
|
||||
print(f"{LOG_PREFIX} {message}", flush=True)
|
||||
|
||||
|
||||
def _run(cmd, cwd=None) -> None:
|
||||
_log("$ " + " ".join(cmd))
|
||||
subprocess.check_call(cmd, cwd=cwd)
|
||||
|
||||
|
||||
def _pip_install(*args: str) -> None:
|
||||
_run([sys.executable, "-m", "pip", "install", *args])
|
||||
|
||||
|
||||
def _git() -> str:
|
||||
git = shutil.which("git")
|
||||
if git is None:
|
||||
raise RuntimeError(
|
||||
"git executable not found on PATH; cannot fetch GitHub dependencies"
|
||||
)
|
||||
return git
|
||||
|
||||
|
||||
def _importable(module_name: str) -> bool:
|
||||
import importlib.util
|
||||
|
||||
try:
|
||||
return importlib.util.find_spec(module_name) is not None
|
||||
except (ImportError, ValueError):
|
||||
return False
|
||||
|
||||
|
||||
def ensure_base_requirements() -> None:
|
||||
"""Install requirements.txt if it clearly has not been installed yet.
|
||||
|
||||
ComfyUI-Manager installs it before running this script, so this only fires
|
||||
for manual `python install.py` users.
|
||||
"""
|
||||
if _importable("diffusers") and _importable("transformers"):
|
||||
_log("base requirements already satisfied")
|
||||
return
|
||||
_pip_install("-r", os.path.join(NODE_DIR, "requirements.txt"))
|
||||
|
||||
|
||||
def ensure_vggt() -> None:
|
||||
"""VGGT (VideoPoseEstimator). Not on PyPI — install from GitHub over https."""
|
||||
if _importable("vggt"):
|
||||
_log("vggt already installed")
|
||||
return
|
||||
_pip_install(f"vggt @ git+{VGGT_GIT_URL}")
|
||||
_log("vggt installed from GitHub")
|
||||
|
||||
|
||||
def ensure_sharp_checkout() -> None:
|
||||
"""Materialize the SHARP submodule (ImageToSplat / VideoToFusedSplats).
|
||||
|
||||
Registry archives already bundle it; git installs need `submodule update`
|
||||
(ComfyUI-Manager does not clone recursively). If this directory is not a
|
||||
git checkout at all, fall back to a direct clone.
|
||||
"""
|
||||
sharp_dir = os.path.join(NODE_DIR, "submodules", "ml-sharpt")
|
||||
sharp_src = os.path.join(sharp_dir, "src", "sharp")
|
||||
if os.path.isdir(sharp_src):
|
||||
_log("SHARP checkout already present")
|
||||
return
|
||||
|
||||
git = _git()
|
||||
if os.path.exists(os.path.join(NODE_DIR, ".git")):
|
||||
_run([git, "submodule", "update", "--init", "--recursive"], cwd=NODE_DIR)
|
||||
if os.path.isdir(sharp_src):
|
||||
_log("SHARP submodule initialized")
|
||||
return
|
||||
|
||||
if os.path.isdir(sharp_dir) and os.listdir(sharp_dir):
|
||||
raise RuntimeError(
|
||||
f"{sharp_dir} exists but does not contain src/sharp; "
|
||||
"remove it and re-run install.py"
|
||||
)
|
||||
_run([git, "clone", "--depth", "1", SHARP_GIT_URL, sharp_dir])
|
||||
_log("SHARP cloned from GitHub")
|
||||
|
||||
|
||||
def ensure_gsplat() -> None:
|
||||
"""gsplat backs SHARP's import chain and the fast splat render path.
|
||||
|
||||
pip install is lightweight (CUDA kernels JIT-compile on first use), but it
|
||||
can still fail on exotic setups — that only disables the SHARP/gsplat nodes.
|
||||
"""
|
||||
if _importable("gsplat"):
|
||||
_log("gsplat already installed")
|
||||
return
|
||||
_pip_install("gsplat")
|
||||
_log("gsplat installed")
|
||||
|
||||
|
||||
def ensure_flux_inpainting_pack() -> None:
|
||||
"""Clone ComfyUI-Flux-Inpainting next to this pack if no copy exists yet."""
|
||||
custom_nodes_dir = os.path.dirname(NODE_DIR)
|
||||
if os.path.basename(custom_nodes_dir).lower() != "custom_nodes":
|
||||
_log(
|
||||
"not installed under a ComfyUI custom_nodes directory; "
|
||||
"skipping ComfyUI-Flux-Inpainting setup"
|
||||
)
|
||||
return
|
||||
|
||||
for name in FLUX_PACK_CANDIDATES:
|
||||
if os.path.isdir(os.path.join(custom_nodes_dir, name)):
|
||||
_log(f"flux inpainting pack already present ({name})")
|
||||
return
|
||||
|
||||
git = _git()
|
||||
target = os.path.join(custom_nodes_dir, "inpainting_flux")
|
||||
_run([git, "clone", "--depth", "1", FLUX_PACK_GIT_URL, target])
|
||||
pack_requirements = os.path.join(target, "requirements.txt")
|
||||
if os.path.isfile(pack_requirements):
|
||||
_pip_install("-r", pack_requirements)
|
||||
_log("ComfyUI-Flux-Inpainting installed as custom_nodes/inpainting_flux")
|
||||
|
||||
|
||||
def ensure_example_inputs() -> None:
|
||||
"""Copy bundled example inputs into ComfyUI's input dir (no overwrite).
|
||||
|
||||
The shipped example workflows reference these files, so a fresh install
|
||||
can queue them immediately.
|
||||
"""
|
||||
src_dir = os.path.join(NODE_DIR, "example_inputs")
|
||||
if not os.path.isdir(src_dir):
|
||||
_log("no example_inputs directory; skipping")
|
||||
return
|
||||
custom_nodes_dir = os.path.dirname(NODE_DIR)
|
||||
if os.path.basename(custom_nodes_dir).lower() != "custom_nodes":
|
||||
_log("not installed under a ComfyUI custom_nodes directory; "
|
||||
"skipping example input setup")
|
||||
return
|
||||
input_dir = os.path.join(os.path.dirname(custom_nodes_dir), "input")
|
||||
os.makedirs(input_dir, exist_ok=True)
|
||||
copied = 0
|
||||
for name in os.listdir(src_dir):
|
||||
target = os.path.join(input_dir, name)
|
||||
if not os.path.exists(target):
|
||||
shutil.copy2(os.path.join(src_dir, name), target)
|
||||
copied += 1
|
||||
_log(f"example inputs ready ({copied} file(s) copied to {input_dir})")
|
||||
|
||||
|
||||
STEPS = (
|
||||
("base requirements", ensure_base_requirements),
|
||||
("vggt (VideoPoseEstimator)", ensure_vggt),
|
||||
("SHARP checkout (ImageToSplat)", ensure_sharp_checkout),
|
||||
("gsplat (splat rasterizer)", ensure_gsplat),
|
||||
("ComfyUI-Flux-Inpainting (OutpaintAnyProjection)", ensure_flux_inpainting_pack),
|
||||
("example workflow inputs", ensure_example_inputs),
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
failures = []
|
||||
for title, step in STEPS:
|
||||
_log(f"--- {title} ---")
|
||||
try:
|
||||
step()
|
||||
except Exception as exc: # keep going: each dependency is optional
|
||||
failures.append((title, exc))
|
||||
_log(f"WARNING: {title} failed: {exc}")
|
||||
|
||||
if failures:
|
||||
_log("finished with warnings - the affected optional nodes stay disabled:")
|
||||
for title, exc in failures:
|
||||
_log(f" * {title}: {exc}")
|
||||
_log("re-run `python install.py` after fixing the issue (network/git/pip)")
|
||||
else:
|
||||
_log("all optional dependencies are ready")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -5,76 +5,107 @@ set -euo pipefail
|
||||
# Functions
|
||||
# ----------------------------------------
|
||||
install_pytorch() {
|
||||
echo "Installing PyTorch, TorchVision, TorchAudio..."
|
||||
echo "==> Installing PyTorch, TorchVision, TorchAudio, bitsandbytes, accelerate…"
|
||||
pip3 install -U torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu128
|
||||
pip3 install bitsandbytes
|
||||
pip3 install accelerate
|
||||
pip3 install -U bitsandbytes accelerate
|
||||
}
|
||||
|
||||
install_system_deps() {
|
||||
echo "Updating apt and installing system dependencies..."
|
||||
echo "==> Updating apt and installing system packages…"
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y build-essential ffmpeg libsm6 libxext6
|
||||
sudo apt-get install -y build-essential ffmpeg libsm6 libxext6 python3.10-dev
|
||||
}
|
||||
|
||||
clone_and_install_comfyui() {
|
||||
echo "Cloning ComfyUI..."
|
||||
echo "==> Cloning ComfyUI…"
|
||||
git clone https://github.com/comfyanonymous/ComfyUI.git
|
||||
echo "Installing ComfyUI requirements..."
|
||||
echo "==> Installing ComfyUI Python requirements…"
|
||||
pip3 install -r ComfyUI/requirements.txt
|
||||
}
|
||||
|
||||
install_camera_node() {
|
||||
echo "Cloning camera‑ComfyUI..."
|
||||
echo "==> Installing camera‑ComfyUI node…"
|
||||
mkdir -p ComfyUI/custom_nodes
|
||||
git clone https://github.com/Alexankharin/camera-comfyUI.git \
|
||||
ComfyUI/custom_nodes/camera-comfyUI
|
||||
echo "Installing camera‑ComfyUI requirements..."
|
||||
pip3 install -r ComfyUI/custom_nodes/camera-comfyUI/requirements.txt
|
||||
# install.py sets up vggt, the SHARP submodule, gsplat and the
|
||||
# inpainting_flux sibling pack (same script ComfyUI-Manager runs).
|
||||
( cd ComfyUI/custom_nodes/camera-comfyUI && python3 install.py )
|
||||
}
|
||||
|
||||
install_image_filters() {
|
||||
echo "Cloning Image‑Filters node..."
|
||||
echo "==> Installing Image‑Filters node…"
|
||||
mkdir -p ComfyUI/custom_nodes
|
||||
git clone https://github.com/spacepxl/ComfyUI-Image-Filters.git \
|
||||
ComfyUI/custom_nodes/ComfyUI-Image-Filters
|
||||
echo "Installing Image‑Filters requirements..."
|
||||
pip3 install -r ComfyUI/custom_nodes/ComfyUI-Image-Filters/requirements.txt
|
||||
}
|
||||
|
||||
clone_flux_inpainting() {
|
||||
echo "Cloning ComfyUI‑Flux‑Inpainting..."
|
||||
echo "==> Installing Flux‑Inpainting node…"
|
||||
mkdir -p ComfyUI/custom_nodes
|
||||
git clone https://github.com/rubi-du/ComfyUI-Flux-Inpainting.git \
|
||||
ComfyUI/custom_nodes/Flux-Inpainting
|
||||
echo "Renaming Flux‑Inpainting folder..."
|
||||
mv ComfyUI/custom_nodes/Flux-Inpainting \
|
||||
ComfyUI/custom_nodes/inpainting_flux
|
||||
ComfyUI/custom_nodes/inpainting_flux
|
||||
}
|
||||
|
||||
install_metric_video_depth_anything() {
|
||||
echo "==> Installing Metric Video Depth Anything…"
|
||||
# Clone into ComfyUI root, not custom_nodes
|
||||
git clone https://github.com/DepthAnything/Video-Depth-Anything.git \
|
||||
ComfyUI/Video-Depth-Anything
|
||||
|
||||
echo " • Installing easydict…"
|
||||
pip3 install -U easydict
|
||||
|
||||
echo " • Copying util.py…"
|
||||
mkdir -p ComfyUI/utils
|
||||
cp ComfyUI/Video-Depth-Anything/metric_depth/utils/util.py \
|
||||
ComfyUI/utils/util.py
|
||||
|
||||
echo " • Downloading Metric Video Depth checkpoint…"
|
||||
mkdir -p ComfyUI/models/checkpoints
|
||||
wget -q -O ComfyUI/models/checkpoints/metric_video_depth_anything_vitl.pth \
|
||||
"https://huggingface.co/depth-anything/Metric-Video-Depth-Anything-Large/resolve/main/metric_video_depth_anything_vitl.pth"
|
||||
}
|
||||
|
||||
|
||||
install_comfyui_manager() {
|
||||
echo "==> Installing ComfyUI-Manager extension…"
|
||||
mkdir -p ComfyUI/custom_nodes
|
||||
git clone https://github.com/Comfy-Org/ComfyUI-Manager.git \
|
||||
ComfyUI/custom_nodes/ComfyUI-Manager
|
||||
pip3 install -r ComfyUI/custom_nodes/ComfyUI-Manager/requirements.txt
|
||||
}
|
||||
|
||||
install_hf_hub() {
|
||||
echo "Installing huggingface_hub..."
|
||||
pip3 install huggingface_hub
|
||||
echo "==> Installing huggingface_hub…"
|
||||
pip3 install -U huggingface_hub
|
||||
}
|
||||
|
||||
download_vae_models() {
|
||||
echo "Downloading WAN‑VACE models via wget..."
|
||||
wget -O ComfyUI/models/vae/wan_2.1_vae.safetensors \
|
||||
echo "==> Downloading WAN‑VACE models…"
|
||||
mkdir -p ComfyUI/models/vae
|
||||
wget -q -O ComfyUI/models/vae/wan_2.1_vae.safetensors \
|
||||
"https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/vae/wan_2.1_vae.safetensors?download=true"
|
||||
|
||||
wget -O ComfyUI/models/text_encoders/umt5_xxl_fp16.safetensors \
|
||||
mkdir -p ComfyUI/models/text_encoders
|
||||
wget -q -O ComfyUI/models/text_encoders/umt5_xxl_fp16.safetensors \
|
||||
"https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/text_encoders/umt5_xxl_fp16.safetensors?download=true"
|
||||
|
||||
wget -O ComfyUI/models/diffusion_models/wan2.1_vace_14B_fp16.safetensors \
|
||||
mkdir -p ComfyUI/models/diffusion_models
|
||||
wget -q -O ComfyUI/models/diffusion_models/wan2.1_vace_14B_fp16.safetensors \
|
||||
"https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/resolve/main/split_files/diffusion_models/wan2.1_vace_14B_fp16.safetensors"
|
||||
}
|
||||
|
||||
login_hf_hub() {
|
||||
echo "Logging in to Hugging Face Hub..."
|
||||
echo "==> Hugging Face login…"
|
||||
huggingface-cli login
|
||||
}
|
||||
|
||||
# ----------------------------------------
|
||||
# Main CLI
|
||||
# Main
|
||||
# ----------------------------------------
|
||||
# default to "install" if no arg given
|
||||
MODE="${1:-install}"
|
||||
|
||||
case "$MODE" in
|
||||
@@ -84,6 +115,7 @@ case "$MODE" in
|
||||
clone_and_install_comfyui
|
||||
install_camera_node
|
||||
install_image_filters
|
||||
install_comfyui_manager
|
||||
install_hf_hub
|
||||
;;
|
||||
|
||||
@@ -95,14 +127,19 @@ case "$MODE" in
|
||||
download_vae_models
|
||||
;;
|
||||
|
||||
depth)
|
||||
install_metric_video_depth_anything
|
||||
;;
|
||||
|
||||
install)
|
||||
install_pytorch
|
||||
install_system_deps
|
||||
clone_and_install_comfyui
|
||||
install_camera_node
|
||||
clone_flux_inpainting
|
||||
install_image_filters
|
||||
install_comfyui_manager
|
||||
install_hf_hub
|
||||
install_metric_video_depth_anything
|
||||
;;
|
||||
|
||||
all)
|
||||
@@ -110,16 +147,17 @@ case "$MODE" in
|
||||
install_system_deps
|
||||
clone_and_install_comfyui
|
||||
install_camera_node
|
||||
clone_flux_inpainting
|
||||
install_image_filters
|
||||
install_comfyui_manager
|
||||
install_hf_hub
|
||||
download_vae_models
|
||||
install_metric_video_depth_anything
|
||||
login_hf_hub
|
||||
echo "All done! 🎉"
|
||||
echo "✅ All done!"
|
||||
;;
|
||||
|
||||
*)
|
||||
echo "Usage: $0 {install|modules|flux|vae|all}"
|
||||
echo "Usage: $0 {install|modules|flux|vae|depth|all}"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
@@ -0,0 +1,287 @@
|
||||
"""Second July-2026 workflow pass: apply the improvements proposed in
|
||||
docs/workflows_review.md.
|
||||
|
||||
1. Point every LoadImage at the bundled example inputs (example_inputs/ is
|
||||
copied into ComfyUI's input dir by install.py), so workflows run on a
|
||||
fresh install without hunting for files.
|
||||
2. video_camera.json dependency trim: drop the dead-end BlurMaskFast
|
||||
(Image-Filters), the easy-mathInt pad computation (Easy-Use) and the now
|
||||
dangling VHS_VideoInfo; swap KJNodes' GetImageRangeFromBatch for the
|
||||
built-in ImageFromBatch. Remaining pack deps: VHS + Florence2.
|
||||
3. Create workflows/record_trajectory.json (TransformToMatrix x2 ->
|
||||
CameraInterpolationNode -> SaveTrajectory) - the missing producer for the
|
||||
trajectory files PC_enricher / wan_vace_ref_to_video consume.
|
||||
4. Stamp `extra.camera_comfyui_rev = 1` in every workflow for future
|
||||
migrations.
|
||||
|
||||
Run: python notebooks/apply_improvements_2026_07.py
|
||||
"""
|
||||
|
||||
import glob
|
||||
import json
|
||||
import os
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
WF = os.path.join(REPO, "workflows")
|
||||
|
||||
PINHOLE = "camera_example_pinhole.jpg"
|
||||
FISHEYE = "camera_example_fisheye.jpg"
|
||||
|
||||
# workflow file -> example image for its LoadImage node(s)
|
||||
LOADIMAGE_DEFAULTS = {
|
||||
"demo_camera_workflow.json": PINHOLE,
|
||||
"Outpaint_node_test.json": PINHOLE,
|
||||
"Outpaint_fisheye180.json": PINHOLE,
|
||||
"outpainting_fisheye_flux.json": PINHOLE,
|
||||
"legacy/outpainting_fisheye.json": PINHOLE,
|
||||
"PointCloud.json": PINHOLE,
|
||||
"pointcloud_walker.json": PINHOLE,
|
||||
"pointcloud_inpaint.json": PINHOLE,
|
||||
"fisheye_to_pointcloud.json": FISHEYE,
|
||||
"legacy/Fisheye_depth_workflow.json": FISHEYE,
|
||||
"PC_enricher.json": FISHEYE,
|
||||
"sbs180_workflow.json": FISHEYE,
|
||||
"wan_vace_ref_to_video.json": FISHEYE,
|
||||
}
|
||||
|
||||
|
||||
def load(rel):
|
||||
with open(os.path.join(WF, rel), encoding="utf-8") as fh:
|
||||
return json.load(fh)
|
||||
|
||||
|
||||
def save(rel, data):
|
||||
with open(os.path.join(WF, rel), "w", encoding="utf-8", newline="\n") as fh:
|
||||
json.dump(data, fh, indent=2, ensure_ascii=False)
|
||||
fh.write("\n")
|
||||
|
||||
|
||||
def node(data, nid):
|
||||
for n in data["nodes"]:
|
||||
if n["id"] == nid:
|
||||
return n
|
||||
raise KeyError(nid)
|
||||
|
||||
|
||||
def set_example_inputs():
|
||||
for rel, image in LOADIMAGE_DEFAULTS.items():
|
||||
d = load(rel)
|
||||
changed = 0
|
||||
for n in d["nodes"]:
|
||||
if n["type"] == "LoadImage":
|
||||
n["widgets_values"][0] = image
|
||||
changed += 1
|
||||
assert changed, f"{rel}: no LoadImage found"
|
||||
save(rel, d)
|
||||
print(f"[ok] {rel}: {changed} LoadImage -> {image}")
|
||||
|
||||
|
||||
def remove_node(data, nid):
|
||||
"""Remove a node and every link touching it; re-index dst slots."""
|
||||
n = node(data, nid)
|
||||
dead = set()
|
||||
for inp in n.get("inputs") or []:
|
||||
if inp.get("link") is not None:
|
||||
dead.add(inp["link"])
|
||||
for out in n.get("outputs") or []:
|
||||
dead.update(out.get("links") or [])
|
||||
data["nodes"] = [x for x in data["nodes"] if x["id"] != nid]
|
||||
data["links"] = [l for l in data["links"] if l[0] not in dead]
|
||||
for x in data["nodes"]:
|
||||
for inp in x.get("inputs") or []:
|
||||
if inp.get("link") in dead:
|
||||
inp["link"] = None
|
||||
for out in x.get("outputs") or []:
|
||||
if out.get("links"):
|
||||
out["links"] = [l for l in out["links"] if l not in dead]
|
||||
|
||||
|
||||
def drop_input(data, nid, name):
|
||||
"""Remove a (widget-converted) input socket and fix slot indices."""
|
||||
n = node(data, nid)
|
||||
inputs = n.get("inputs") or []
|
||||
n["inputs"] = [i for i in inputs if i["name"] != name]
|
||||
index = {i["name"]: k for k, i in enumerate(n["inputs"])}
|
||||
for l in data["links"]:
|
||||
if l[3] == nid:
|
||||
# find by link id which input holds it
|
||||
for i in n["inputs"]:
|
||||
if i.get("link") == l[0]:
|
||||
l[4] = index[i["name"]]
|
||||
|
||||
|
||||
def bbox(data, ids, pad_top=60, pad=20):
|
||||
xs, ys, xe, ye = [], [], [], []
|
||||
for nid in ids:
|
||||
n = node(data, nid)
|
||||
p, s = n["pos"], n["size"]
|
||||
xs.append(p[0]); ys.append(p[1])
|
||||
xe.append(p[0] + s[0]); ye.append(p[1] + s[1])
|
||||
x, y = min(xs) - pad, min(ys) - pad_top
|
||||
return [x, y, max(xe) + pad - x, max(ye) + pad - y]
|
||||
|
||||
|
||||
def fix_video_camera(rel="video_camera.json"):
|
||||
d = load(rel)
|
||||
# 1. dead-end mask blur (only Image-Filters usage; radius was 0/0 anyway)
|
||||
remove_node(d, 60)
|
||||
# 2. pad computation chain (Easy-Use mathInt x2 + VHS_VideoInfo feeding it);
|
||||
# ImagePadForOutpaint falls back to its stored widgets (280/280)
|
||||
for nid in (27, 26, 23):
|
||||
remove_node(d, nid)
|
||||
for name in ("top", "bottom"):
|
||||
drop_input(d, 21, name)
|
||||
n21 = node(d, 21)
|
||||
n21["title"] = "Pad to square — set top/bottom to (W−H)/2"
|
||||
# 3. KJNodes GetImageRangeFromBatch -> built-in ImageFromBatch (frame 0)
|
||||
n63 = node(d, 63)
|
||||
n63["type"] = "ImageFromBatch"
|
||||
n63["properties"]["Node name for S&R"] = "ImageFromBatch"
|
||||
n63["inputs"][0]["name"] = "image"
|
||||
n63["widgets_values"] = [0, 1] # batch_index, length
|
||||
n63["title"] = "First frame (for captioning)"
|
||||
# regroup without the removed nodes
|
||||
groups = [
|
||||
("1. Load & pad video", [9, 21, 68]),
|
||||
("2. Metric video depth", [19, 12]),
|
||||
("3. Re-render with new camera", [16, 11, 10, 6]),
|
||||
("4. Masks & composite", [39, 41, 75, 54, 74]),
|
||||
("5. Auto-caption (Florence2)", [63, 62, 61]),
|
||||
("6. WAN VACE re-generation", [29, 30, 37, 32, 33, 34, 28, 31, 36, 35]),
|
||||
("7. Outputs", [4, 13, 14, 38]),
|
||||
]
|
||||
colors = ["#3f789e", "#a1309b", "#8A8", "#b58b2a", "#88A", "#b06634", "#535"]
|
||||
d["groups"] = [{
|
||||
"id": i + 1, "title": t, "bounding": bbox(d, ids),
|
||||
"color": colors[i % len(colors)], "font_size": 24, "flags": {},
|
||||
} for i, (t, ids) in enumerate(groups)]
|
||||
# refresh the note (dependency list changed)
|
||||
for n in d["nodes"]:
|
||||
if n["type"] == "MarkdownNote" and n.get("title") == "About this workflow":
|
||||
n["widgets_values"] = [(
|
||||
"# Re-shoot a video with a new camera move\n\n"
|
||||
"Re-renders an input video along a user-defined camera "
|
||||
"trajectory and uses WAN 2.1 VACE to regenerate what the new "
|
||||
"camera reveals:\n\n"
|
||||
"1. Video is padded square (ImagePadForOutpaint — set "
|
||||
"top/bottom to (width−height)/2 for your video) and "
|
||||
"depth-estimated per frame (Video-Depth-Anything metric).\n"
|
||||
"2. **VideoCameraMotionSequence** lifts each frame to a point "
|
||||
"cloud and re-renders it along the SE(3)-interpolated "
|
||||
"trajectory.\n"
|
||||
"3. Disocclusion masks + re-rendered frames become VACE "
|
||||
"control video/masks; **Florence2** auto-captions the clip as "
|
||||
"the prompt; WAN 2.1 VACE 14B fills the gaps.\n\n"
|
||||
"- **Set:** video path (VHS Load Video Path); the camera move "
|
||||
"(two TransformToMatrix poses); pad amounts; override the "
|
||||
"auto-caption in CLIPTextEncode if desired.\n"
|
||||
"- **Requires (packs):** VideoHelperSuite, ComfyUI-Florence2.\n"
|
||||
"- **Requires (models):** `metric_video_depth_anything_vitl"
|
||||
".pth` (install.sh `depth`), WAN 2.1 VACE 14B + umt5-xxl + "
|
||||
"WAN VAE (install.sh `vae`), Florence-2 (auto-download).\n"
|
||||
"- **Outputs:** re-rendered composite WEBM, depth WEBM, final "
|
||||
"VACE clip."
|
||||
)]
|
||||
save(rel, d)
|
||||
print(f"[ok] {rel}: removed BlurMaskFast/mathInt/VideoInfo, "
|
||||
f"ImageFromBatch swap, regrouped")
|
||||
|
||||
|
||||
def make_record_trajectory(rel="record_trajectory.json"):
|
||||
note = (
|
||||
"# Record a camera trajectory\n\n"
|
||||
"Produces the `.npy` trajectory file consumed by `PC_enricher.json` "
|
||||
"and `wan_vace_ref_to_video.json` (LoadTrajectory): two poses are "
|
||||
"SE(3)-interpolated into a smooth 20-step path and saved by "
|
||||
"**SaveTrajectory** to your ComfyUI **output** directory.\n\n"
|
||||
"- **Set:** the end pose (shift XYZ in scene units — metric if the "
|
||||
"cloud came from metric depth — plus theta = pitch, phi = yaw) and "
|
||||
"`num_steps`.\n"
|
||||
"- **Then:** move the saved file from `output/` to `input/` so "
|
||||
"LoadTrajectory can list it. A bundled example "
|
||||
"(`ComfyUITrajectory_00001.npy`) is already installed by install.py.\n"
|
||||
"- Chain more CameraInterpolationNode segments (or use "
|
||||
"CameraTrajectoryNode on a point cloud) for multi-keyframe paths."
|
||||
)
|
||||
d = {
|
||||
"id": "00000000-0000-0000-0000-000000000000",
|
||||
"revision": 0,
|
||||
"last_node_id": 5,
|
||||
"last_link_id": 3,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 1, "type": "TransformToMatrix", "title": "Start pose (identity)",
|
||||
"pos": [-500, 320], "size": [315, 154], "flags": {}, "order": 0,
|
||||
"mode": 0, "inputs": [],
|
||||
"outputs": [{"name": "transformation matrix", "type": "MAT_4X4",
|
||||
"links": [1], "slot_index": 0}],
|
||||
"properties": {"Node name for S&R": "TransformToMatrix"},
|
||||
"widgets_values": [0.0, 0.0, 0.0, 0.0, 0.0],
|
||||
},
|
||||
{
|
||||
"id": 2, "type": "TransformToMatrix", "title": "End pose (edit me)",
|
||||
"pos": [-500, 540], "size": [315, 154], "flags": {}, "order": 1,
|
||||
"mode": 0, "inputs": [],
|
||||
"outputs": [{"name": "transformation matrix", "type": "MAT_4X4",
|
||||
"links": [2], "slot_index": 0}],
|
||||
"properties": {"Node name for S&R": "TransformToMatrix"},
|
||||
"widgets_values": [0.0, 0.0, 0.3, 0.0, 30.0],
|
||||
},
|
||||
{
|
||||
"id": 3, "type": "CameraInterpolationNode",
|
||||
"title": "Interpolate 20 poses",
|
||||
"pos": [-120, 430], "size": [226, 78], "flags": {}, "order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{"name": "initial_matrix", "type": "MAT_4X4", "link": 1},
|
||||
{"name": "final_matrix", "type": "MAT_4X4", "link": 2},
|
||||
],
|
||||
"outputs": [{"name": "trajectory", "type": "TENSOR",
|
||||
"links": [3], "slot_index": 0}],
|
||||
"properties": {"Node name for S&R": "CameraInterpolationNode"},
|
||||
"widgets_values": [20],
|
||||
},
|
||||
{
|
||||
"id": 4, "type": "SaveTrajectory", "title": "Save to output/*.npy",
|
||||
"pos": [170, 430], "size": [315, 82], "flags": {}, "order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [{"name": "trajectory", "type": "TENSOR", "link": 3}],
|
||||
"outputs": [],
|
||||
"properties": {"Node name for S&R": "SaveTrajectory"},
|
||||
"widgets_values": ["ComfyUITrajectory"],
|
||||
},
|
||||
{
|
||||
"id": 5, "type": "MarkdownNote", "title": "About this workflow",
|
||||
"pos": [-1080, 320], "size": [520, 430], "flags": {}, "order": 4,
|
||||
"mode": 0, "inputs": [], "outputs": [], "properties": {},
|
||||
"widgets_values": [note], "color": "#432", "bgcolor": "#653",
|
||||
},
|
||||
],
|
||||
"links": [
|
||||
[1, 1, 0, 3, 0, "MAT_4X4"],
|
||||
[2, 2, 0, 3, 1, "MAT_4X4"],
|
||||
[3, 3, 0, 4, 0, "TENSOR"],
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {},
|
||||
"version": 0.4,
|
||||
}
|
||||
save(rel, d)
|
||||
print(f"[ok] created {rel}")
|
||||
|
||||
|
||||
def stamp_revision():
|
||||
for path in glob.glob(os.path.join(WF, "**", "*.json"), recursive=True):
|
||||
rel = os.path.relpath(path, WF)
|
||||
d = load(rel)
|
||||
d.setdefault("extra", {})["camera_comfyui_rev"] = 1
|
||||
save(rel, d)
|
||||
print("[ok] stamped extra.camera_comfyui_rev = 1")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
set_example_inputs()
|
||||
fix_video_camera()
|
||||
make_record_trajectory()
|
||||
stamp_revision()
|
||||
@@ -0,0 +1,294 @@
|
||||
"""Build workflows/fisheye_static_video_to_4d.json.
|
||||
|
||||
Static-fisheye-camera variant of video_to_4d_world.json: because the camera
|
||||
does not move, no VGGT pose estimation is needed — the trajectory is identity,
|
||||
per-frame depth comes from the batched FisheyeDepthEstimator, and the whole
|
||||
static background is a single FisheyeToGaussian prediction of frame 0 split by
|
||||
the motion mask (outside = static world, inside = dynamic canonical).
|
||||
|
||||
Node facts this graph relies on (verified against current INPUT_TYPES):
|
||||
- FisheyeDepthEstimator is batched: IMAGE [T,H,W,C] -> depthmap [T,H,W,1];
|
||||
GS4D nodes squeeze the trailing channel (GS4D_nodes.py::167).
|
||||
- MotionMaskFromDepth interpolates a [K,4,4] trajectory to T internally.
|
||||
- TracksToTrajectories' trajectory input is optional and defaults to identity
|
||||
("static camera" per its tooltip) — left unconnected on purpose.
|
||||
- SplitSplatsByMask returns (inside_splats, outside_splats); its optional
|
||||
camera_matrix defaults to identity, which is exactly the frame-0 camera here.
|
||||
|
||||
Run: python notebooks/build_fisheye_static_4d.py
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
OUT = os.path.join(REPO, "workflows", "fisheye_static_video_to_4d.json")
|
||||
|
||||
NODES = []
|
||||
LINKS = []
|
||||
_link_id = 0
|
||||
|
||||
|
||||
def node(nid, ntype, title, pos, size, widgets, inputs=(), outputs=(), mode=0,
|
||||
extra=None):
|
||||
n = {
|
||||
"id": nid, "type": ntype, "pos": list(pos), "size": list(size),
|
||||
"flags": {}, "order": len(NODES), "mode": mode,
|
||||
"inputs": [dict(i) for i in inputs],
|
||||
"outputs": [
|
||||
{"name": o[0], "type": o[1], "links": [], "slot_index": k}
|
||||
for k, o in enumerate(outputs)
|
||||
],
|
||||
"properties": {"Node name for S&R": ntype},
|
||||
"widgets_values": widgets,
|
||||
}
|
||||
if title:
|
||||
n["title"] = title
|
||||
if extra:
|
||||
n.update(extra)
|
||||
NODES.append(n)
|
||||
return n
|
||||
|
||||
|
||||
def link(src, src_slot, dst, dst_input, ltype):
|
||||
global _link_id
|
||||
_link_id += 1
|
||||
sn = next(n for n in NODES if n["id"] == src)
|
||||
dn = next(n for n in NODES if n["id"] == dst)
|
||||
slot = next(i for i, inp in enumerate(dn["inputs"]) if inp["name"] == dst_input)
|
||||
dn["inputs"][slot]["link"] = _link_id
|
||||
sn["outputs"][src_slot]["links"].append(_link_id)
|
||||
LINKS.append([_link_id, src, src_slot, dst, slot, ltype])
|
||||
|
||||
|
||||
def inp(name, ltype):
|
||||
return {"name": name, "type": ltype, "link": None}
|
||||
|
||||
|
||||
NOTE = """# Static fisheye video → 4D Gaussian video
|
||||
|
||||
Turns a video shot on a **static 180° fisheye camera** (locked-off shot,
|
||||
security cam, tripod) into a 4D (3D + time) Gaussian scene, then re-renders it
|
||||
from a *moving* novel camera.
|
||||
|
||||
A static camera needs **no pose estimation** (no VGGT, unlike
|
||||
`video_to_4d_world.json`): the trajectory is identity, and one frame already
|
||||
sees the entire static background.
|
||||
|
||||
**Stages**
|
||||
1. **FisheyeDepthEstimator** — per-frame metric *radial* depth on the whole
|
||||
fisheye batch (DISTANCE_AWARE multi-view merge).
|
||||
2. **MotionMaskFromDepth** (identity trajectory) — pixels whose depth changes
|
||||
over time are dynamic.
|
||||
3. **FisheyeToGaussian** on frame 0 → whole-scene splats;
|
||||
**SplitSplatsByMask** (FISHEYE 180°) separates the dynamic canonical
|
||||
(inside) from the static world (outside).
|
||||
4. **EstimateTracks** (CoTracker3) + **TracksToTrajectories** (FISHEYE 180°,
|
||||
identity poses — its default) lift 2D tracks to 3D control trajectories.
|
||||
5. **BuildSplats4D** binds the canonical splats to the tracks → 4D scene with
|
||||
the static background attached.
|
||||
6. **RenderSplats4DVideo** renders a novel orbit (pinhole 90°). A second,
|
||||
**muted** render replays the original static fisheye view for A/B
|
||||
comparison — unmute it (Ctrl+M) to use.
|
||||
|
||||
**Set:** the video path + `frame_load_cap` (≤ 64 recommended); the novel-path
|
||||
end pose; `threshold` in MotionMaskFromDepth (raise it if the mask flickers —
|
||||
per-frame depth is not temporally consistent).
|
||||
|
||||
**Requires:** SHARP + gsplat (auto via install.py), CoTracker3 (torch.hub,
|
||||
first-use download), Depth-Anything V2 (auto), VideoHelperSuite pack.
|
||||
|
||||
**Limitations:** the static world only knows what frame 0 saw — regions
|
||||
occluded by moving objects at t=0 are holes if the novel camera peeks behind
|
||||
them; strong depth flicker can leak static pixels into the dynamic set.
|
||||
|
||||
**Outputs:** novel-path WEBM, the 4D scene as `.npz` (SaveSplats4D — reload
|
||||
with LoadSplats4D), optional replay WEBM."""
|
||||
|
||||
DA = "Depth-Anything-V2-Metric-Indoor-Base-hf"
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
node(30, "MarkdownNote", "About this workflow", (-2000, 260), (560, 700), [NOTE],
|
||||
extra={"color": "#432", "bgcolor": "#653"})
|
||||
|
||||
node(1, "VHS_LoadVideoPath", "Load fisheye video (static camera, <= 64 frames)",
|
||||
(-1380, 300), (240, 262),
|
||||
{
|
||||
"video": "input/fisheye_video.mp4",
|
||||
"force_rate": 0, "custom_width": 0, "custom_height": 0,
|
||||
"frame_load_cap": 49, "skip_first_frames": 0, "select_every_nth": 1,
|
||||
"format": "AnimateDiff",
|
||||
"videopreview": {"hidden": False, "paused": False, "params": {
|
||||
"filename": "input/fisheye_video.mp4", "type": "path",
|
||||
"format": "video/mp4", "force_rate": 0, "custom_width": 0,
|
||||
"custom_height": 0, "frame_load_cap": 49,
|
||||
"skip_first_frames": 0, "select_every_nth": 1}},
|
||||
},
|
||||
inputs=[inp("meta_batch", "VHS_BatchManager"), inp("vae", "VAE")],
|
||||
outputs=[("IMAGE", "IMAGE"), ("frame_count", "INT"),
|
||||
("audio", "AUDIO"), ("video_info", "VHS_VIDEOINFO")])
|
||||
|
||||
node(2, "FisheyeDepthEstimator", "Per-frame radial depth (batched)",
|
||||
(-1060, 300), (315, 246),
|
||||
[DA, 1.0, 90.0, 518, 1024, "DISTANCE_AWARE", 25, 1],
|
||||
inputs=[inp("image", "IMAGE")],
|
||||
outputs=[("depthmap", "TENSOR"), ("mask", "MASK")])
|
||||
|
||||
node(3, "TransformToMatrix", "Static camera (identity pose)",
|
||||
(-1060, 640), (315, 154), [0.0, 0.0, 0.0, 0.0, 0.0],
|
||||
outputs=[("transformation matrix", "MAT_4X4")])
|
||||
|
||||
node(4, "CameraInterpolationNode", "Identity trajectory (interpolated to T)",
|
||||
(-680, 680), (226, 78), [2],
|
||||
inputs=[inp("initial_matrix", "MAT_4X4"), inp("final_matrix", "MAT_4X4")],
|
||||
outputs=[("trajectory", "TENSOR")])
|
||||
|
||||
node(5, "MotionMaskFromDepth", "Dynamic-pixel mask (1 = moving)",
|
||||
(-680, 300), (315, 202), ["FISHEYE", 180.0, 0.15, 4, 2, "auto"],
|
||||
inputs=[inp("depth_seq", "TENSOR"), inp("trajectory", "TENSOR")],
|
||||
outputs=[("motion_mask", "MASK")])
|
||||
|
||||
node(6, "ImageFromBatch", "Frame 0 (canonical view)",
|
||||
(-680, 560), (226, 82), [0, 1],
|
||||
inputs=[inp("image", "IMAGE")],
|
||||
outputs=[("IMAGE", "IMAGE")])
|
||||
|
||||
node(7, "FisheyeToGaussian", "SHARP frame 0 -> whole-scene splats",
|
||||
(-300, 460), (330, 290),
|
||||
[180.0, 0, 0, "<download default>", "auto", 90.0, 0, "smart", 0.01, 5.0],
|
||||
inputs=[inp("image", "IMAGE")],
|
||||
outputs=[("splats", "GSPLAT")])
|
||||
|
||||
node(8, "EstimateTracks", "CoTracker3 (downloads on first use)",
|
||||
(-300, 820), (300, 102), [20, "auto"],
|
||||
inputs=[inp("frames", "IMAGE")],
|
||||
outputs=[("tracks", "TENSOR"), ("visibility", "TENSOR")])
|
||||
|
||||
node(9, "SplitSplatsByMask", "Split: inside = dynamic, outside = static world",
|
||||
(100, 300), (315, 174), ["FISHEYE", 180.0, 0.5, "auto"],
|
||||
inputs=[inp("splats", "GSPLAT"), inp("mask", "MASK"),
|
||||
inp("camera_matrix", "MAT_4X4")],
|
||||
outputs=[("inside_splats", "GSPLAT"), ("outside_splats", "GSPLAT")])
|
||||
|
||||
node(10, "TracksToTrajectories", "Lift tracks to 3D (identity poses = default)",
|
||||
(100, 700), (315, 190), ["FISHEYE", 180.0, 0.5, "auto"],
|
||||
inputs=[inp("tracks", "TENSOR"), inp("visibility", "TENSOR"),
|
||||
inp("depth_seq", "TENSOR"), inp("trajectory", "TENSOR")],
|
||||
outputs=[("trajectories3d", "TENSOR"), ("track_valid", "TENSOR")])
|
||||
|
||||
node(11, "BuildSplats4D", "Bind canonical to tracks -> 4D scene",
|
||||
(500, 440), (315, 190), [0, 4, 0.0, "auto"],
|
||||
inputs=[inp("canonical", "GSPLAT"), inp("trajectories3d", "TENSOR"),
|
||||
inp("static", "GSPLAT"), inp("times", "TENSOR"),
|
||||
inp("track_valid", "TENSOR")],
|
||||
outputs=[("splats4d", "GSPLAT4D")])
|
||||
|
||||
node(12, "TransformToMatrix", "Novel path: start (original camera)",
|
||||
(500, 720), (315, 154), [0.0, 0.0, 0.0, 0.0, 0.0],
|
||||
outputs=[("transformation matrix", "MAT_4X4")])
|
||||
|
||||
node(13, "TransformToMatrix", "Novel path: end (small orbit — edit me)",
|
||||
(500, 920), (315, 154), [0.15, 0.0, 0.1, 0.0, -10.0],
|
||||
outputs=[("transformation matrix", "MAT_4X4")])
|
||||
|
||||
node(14, "CameraInterpolationNode", "Novel camera path",
|
||||
(880, 820), (226, 78), [2],
|
||||
inputs=[inp("initial_matrix", "MAT_4X4"), inp("final_matrix", "MAT_4X4")],
|
||||
outputs=[("trajectory", "TENSOR")])
|
||||
|
||||
node(15, "RenderSplats4DVideo", "Render 4D along novel path (pinhole 90°)",
|
||||
(900, 300), (315, 266), [49, 0.0, 1.0, "PINHOLE", 90.0, 768, 768, "auto", 0, "auto"],
|
||||
inputs=[inp("splats4d", "GSPLAT4D"), inp("trajectory", "TENSOR")],
|
||||
outputs=[("images", "IMAGE"), ("masks", "MASK"), ("disparity", "TENSOR")])
|
||||
|
||||
node(16, "RenderSplats4DVideo", "MUTED: replay original fisheye view (A/B check)",
|
||||
(900, 1060), (315, 266), [49, 0.0, 1.0, "FISHEYE", 180.0, 1024, 1024, "auto", 0, "auto"],
|
||||
inputs=[inp("splats4d", "GSPLAT4D"), inp("trajectory", "TENSOR")],
|
||||
outputs=[("images", "IMAGE"), ("masks", "MASK"), ("disparity", "TENSOR")],
|
||||
mode=2)
|
||||
|
||||
node(17, "SaveWEBM", None, (1300, 300), (315, 437),
|
||||
["4d_fisheye_novel", "vp9", 24, 32],
|
||||
inputs=[inp("images", "IMAGE")])
|
||||
|
||||
node(18, "SaveSplats4D", "Save 4D scene (.npz)", (1300, 800), (315, 106),
|
||||
["ComfyUISplat4D_fisheye", False],
|
||||
inputs=[inp("splats4d", "GSPLAT4D")])
|
||||
|
||||
node(19, "SaveWEBM", "MUTED: replay output", (1300, 1060), (315, 437),
|
||||
["4d_fisheye_replay", "vp9", 24, 32],
|
||||
inputs=[inp("images", "IMAGE")], mode=2)
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
link(1, 0, 2, "image", "IMAGE")
|
||||
link(1, 0, 6, "image", "IMAGE")
|
||||
link(1, 0, 8, "frames", "IMAGE")
|
||||
link(2, 0, 5, "depth_seq", "TENSOR")
|
||||
link(2, 0, 10, "depth_seq", "TENSOR")
|
||||
link(3, 0, 4, "initial_matrix", "MAT_4X4")
|
||||
link(3, 0, 4, "final_matrix", "MAT_4X4")
|
||||
link(4, 0, 5, "trajectory", "TENSOR")
|
||||
link(4, 0, 16, "trajectory", "TENSOR")
|
||||
link(5, 0, 9, "mask", "MASK")
|
||||
link(6, 0, 7, "image", "IMAGE")
|
||||
link(7, 0, 9, "splats", "GSPLAT")
|
||||
link(8, 0, 10, "tracks", "TENSOR")
|
||||
link(8, 1, 10, "visibility", "TENSOR")
|
||||
link(9, 0, 11, "canonical", "GSPLAT")
|
||||
link(9, 1, 11, "static", "GSPLAT")
|
||||
link(10, 0, 11, "trajectories3d", "TENSOR")
|
||||
link(10, 1, 11, "track_valid", "TENSOR")
|
||||
link(11, 0, 15, "splats4d", "GSPLAT4D")
|
||||
link(11, 0, 16, "splats4d", "GSPLAT4D")
|
||||
link(11, 0, 18, "splats4d", "GSPLAT4D")
|
||||
link(12, 0, 14, "initial_matrix", "MAT_4X4")
|
||||
link(13, 0, 14, "final_matrix", "MAT_4X4")
|
||||
link(14, 0, 15, "trajectory", "TENSOR")
|
||||
link(15, 0, 17, "images", "IMAGE")
|
||||
link(16, 0, 19, "images", "IMAGE")
|
||||
# NOTE: TracksToTrajectories.trajectory stays unconnected on purpose — its
|
||||
# default is identity poses, i.e. exactly the static camera.
|
||||
|
||||
|
||||
def bbox(ids, pad_top=60, pad=20):
|
||||
ns = [n for n in NODES if n["id"] in ids]
|
||||
x = min(n["pos"][0] for n in ns) - pad
|
||||
y = min(n["pos"][1] for n in ns) - pad_top
|
||||
x2 = max(n["pos"][0] + n["size"][0] for n in ns) + pad
|
||||
y2 = max(n["pos"][1] + n["size"][1] for n in ns) + pad
|
||||
return [x, y, x2 - x, y2 - y]
|
||||
|
||||
|
||||
GROUPS = [
|
||||
("1. Load fisheye video", [1]),
|
||||
("2. Per-frame fisheye depth", [2]),
|
||||
("3. Static camera trajectory", [3, 4]),
|
||||
("4. Motion mask", [5]),
|
||||
("5. Frame-0 splats & static/dynamic split", [6, 7, 9]),
|
||||
("6. Dynamic tracks -> 3D", [8, 10]),
|
||||
("7. 4D scene", [11, 18]),
|
||||
("8. Render novel path (+ muted replay)", [12, 13, 14, 15, 16, 17, 19]),
|
||||
]
|
||||
COLORS = ["#3f789e", "#a1309b", "#8A8", "#b58b2a", "#88A", "#b06634", "#535",
|
||||
"#3f789e"]
|
||||
|
||||
workflow = {
|
||||
"id": "00000000-0000-0000-0000-000000000000",
|
||||
"revision": 0,
|
||||
"last_node_id": 30,
|
||||
"last_link_id": _link_id,
|
||||
"nodes": NODES,
|
||||
"links": LINKS,
|
||||
"groups": [{
|
||||
"id": i + 1, "title": t, "bounding": bbox(ids),
|
||||
"color": COLORS[i % len(COLORS)], "font_size": 24, "flags": {},
|
||||
} for i, (t, ids) in enumerate(GROUPS)],
|
||||
"config": {},
|
||||
"extra": {"camera_comfyui_rev": 1},
|
||||
"version": 0.4,
|
||||
}
|
||||
|
||||
with open(OUT, "w", encoding="utf-8", newline="\n") as fh:
|
||||
json.dump(workflow, fh, indent=2, ensure_ascii=False)
|
||||
fh.write("\n")
|
||||
print(f"wrote {OUT}: {len(NODES)} nodes, {len(LINKS)} links")
|
||||
@@ -0,0 +1,699 @@
|
||||
"""One-shot July 2026 workflow rework: schema repairs + documentation.
|
||||
|
||||
For every workflows/*.json this script:
|
||||
1. migrates widgets_values stored with pre-2026 node schemas to the current
|
||||
widget lists (see notebooks/validate_workflows.py for the detector);
|
||||
2. fixes broken references (wan UNET filename typo, ReprojectImage rotation
|
||||
widgets that no longer exist -> explicit TransformToMatrix nodes);
|
||||
3. adds an embedded MarkdownNote explaining purpose/stages/requirements,
|
||||
meaningful group boxes and node titles;
|
||||
4. rewrites the JSON pretty-printed (indent 2).
|
||||
|
||||
Idempotent-ish: repairs are guarded by length/value checks, notes are only
|
||||
added if no MarkdownNote titled 'About this workflow' exists.
|
||||
|
||||
Run: python notebooks/rework_workflows_2026_07.py
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
WF = os.path.join(REPO, "workflows")
|
||||
|
||||
DA_MODEL = "Depth-Anything-V2-Metric-Indoor-Base-hf"
|
||||
|
||||
NOTE_TITLE = "About this workflow"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Generic helpers
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def load(name):
|
||||
with open(os.path.join(WF, name), encoding="utf-8") as fh:
|
||||
return json.load(fh)
|
||||
|
||||
|
||||
def save(name, data):
|
||||
# normalize pos/size dict-form (pre-2024 serialization) to arrays
|
||||
for n in data.get("nodes", []):
|
||||
for key in ("pos", "size"):
|
||||
v = n.get(key)
|
||||
if isinstance(v, dict):
|
||||
n[key] = [v[k] for k in sorted(v, key=lambda s: int(s))]
|
||||
with open(os.path.join(WF, name), "w", encoding="utf-8", newline="\n") as fh:
|
||||
json.dump(data, fh, indent=2, ensure_ascii=False)
|
||||
fh.write("\n")
|
||||
|
||||
|
||||
def node(data, nid):
|
||||
for n in data["nodes"]:
|
||||
if n["id"] == nid:
|
||||
return n
|
||||
raise KeyError(f"node {nid} not found")
|
||||
|
||||
|
||||
def xy(v):
|
||||
if isinstance(v, dict):
|
||||
return [float(v["0"]), float(v["1"])]
|
||||
return [float(v[0]), float(v[1])]
|
||||
|
||||
|
||||
def bbox(data, ids, pad_top=60, pad=20):
|
||||
xs, ys, xe, ye = [], [], [], []
|
||||
for nid in ids:
|
||||
n = node(data, nid)
|
||||
p, s = xy(n["pos"]), xy(n["size"])
|
||||
xs.append(p[0]); ys.append(p[1])
|
||||
xe.append(p[0] + s[0]); ye.append(p[1] + s[1])
|
||||
x, y = min(xs) - pad, min(ys) - pad_top
|
||||
return [x, y, max(xe) + pad - x, max(ye) + pad - y]
|
||||
|
||||
|
||||
GROUP_COLORS = ["#3f789e", "#a1309b", "#8A8", "#b58b2a", "#88A", "#b06634", "#535"]
|
||||
|
||||
|
||||
def set_groups(data, groups):
|
||||
"""groups: list of (title, node_ids). Replaces the groups list."""
|
||||
out = []
|
||||
for i, (title, ids) in enumerate(groups):
|
||||
out.append({
|
||||
"id": i + 1,
|
||||
"title": title,
|
||||
"bounding": bbox(data, ids),
|
||||
"color": GROUP_COLORS[i % len(GROUP_COLORS)],
|
||||
"font_size": 24,
|
||||
"flags": {},
|
||||
})
|
||||
data["groups"] = out
|
||||
|
||||
|
||||
def retitle_groups(data, titles):
|
||||
"""titles: list matching data['groups'] order."""
|
||||
assert len(titles) == len(data.get("groups", [])), "group count mismatch"
|
||||
for g, t in zip(data["groups"], titles):
|
||||
g["title"] = t
|
||||
|
||||
|
||||
def set_title(data, nid, title):
|
||||
node(data, nid)["title"] = title
|
||||
|
||||
|
||||
def next_node_id(data):
|
||||
data["last_node_id"] = int(data.get("last_node_id", 0)) + 1
|
||||
return data["last_node_id"]
|
||||
|
||||
|
||||
def next_link_id(data):
|
||||
data["last_link_id"] = int(data.get("last_link_id", 0)) + 1
|
||||
return data["last_link_id"]
|
||||
|
||||
|
||||
def add_note(data, markdown, pos=None, size=None):
|
||||
for n in data["nodes"]:
|
||||
if n["type"] in ("MarkdownNote", "Note") and n.get("title") == NOTE_TITLE:
|
||||
n["widgets_values"] = [markdown]
|
||||
return n["id"]
|
||||
if pos is None:
|
||||
xs = [xy(n["pos"])[0] for n in data["nodes"]]
|
||||
ys = [xy(n["pos"])[1] for n in data["nodes"]]
|
||||
pos = [min(xs) - 560, min(ys)]
|
||||
if size is None:
|
||||
lines = markdown.count("\n") + 1
|
||||
size = [520, max(220, min(700, 26 * lines + 80))]
|
||||
nid = next_node_id(data)
|
||||
data["nodes"].append({
|
||||
"id": nid,
|
||||
"type": "MarkdownNote",
|
||||
"title": NOTE_TITLE,
|
||||
"pos": pos,
|
||||
"size": size,
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [markdown],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653",
|
||||
})
|
||||
return nid
|
||||
|
||||
|
||||
def add_transform_node(data, pos, widgets):
|
||||
nid = next_node_id(data)
|
||||
data["nodes"].append({
|
||||
"id": nid,
|
||||
"type": "TransformToMatrix",
|
||||
"pos": pos,
|
||||
"size": [315, 154],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [{"name": "transformation matrix", "type": "MAT_4X4",
|
||||
"links": [], "slot_index": 0}],
|
||||
"properties": {"Node name for S&R": "TransformToMatrix"},
|
||||
"widgets_values": widgets,
|
||||
})
|
||||
return nid
|
||||
|
||||
|
||||
def link(data, src, dst, dst_input, ltype="MAT_4X4", src_slot=0):
|
||||
"""Connect src node output slot to dst node's named input (added if absent)."""
|
||||
dnode = node(data, dst)
|
||||
inputs = dnode.setdefault("inputs", [])
|
||||
slot = None
|
||||
for i, inp in enumerate(inputs):
|
||||
if inp["name"] == dst_input:
|
||||
slot = i
|
||||
break
|
||||
if slot is None:
|
||||
inputs.append({"name": dst_input, "type": ltype, "link": None})
|
||||
slot = len(inputs) - 1
|
||||
lid = next_link_id(data)
|
||||
inputs[slot]["link"] = lid
|
||||
snode = node(data, src)
|
||||
snode["outputs"][src_slot].setdefault("links", None)
|
||||
if snode["outputs"][src_slot]["links"] is None:
|
||||
snode["outputs"][src_slot]["links"] = []
|
||||
snode["outputs"][src_slot]["links"].append(lid)
|
||||
data["links"].append([lid, src, src_slot, dst, slot, ltype])
|
||||
return lid
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Global widget-schema repairs (2025 -> 2026 node schemas)
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def repair_widgets(data):
|
||||
changed = []
|
||||
for n in data["nodes"]:
|
||||
t, w = n["type"], n.get("widgets_values")
|
||||
if not isinstance(w, list):
|
||||
continue
|
||||
if t == "DepthEstimatorNode" and len(w) == 2:
|
||||
n["widgets_values"] = w + [1] # median_blur_kernel
|
||||
elif t == "FisheyeDepthEstimator" and len(w) == 6:
|
||||
# old: [model, scale, pfov, pres, fres, softmerge_radius]
|
||||
n["widgets_values"] = w[:5] + ["SOFTMERGE", w[5], 1]
|
||||
elif t == "PointCloudCleaner" and len(w) == 2:
|
||||
# old (voxel_size, min_points) were world-unit; semantics changed
|
||||
n["widgets_values"] = [1024, 1024, 1.0, 3]
|
||||
elif t == "CameraMotionNode" and len(w) == 6:
|
||||
n["widgets_values"] = w + [0, False, False]
|
||||
elif t == "CameraInterpolationNode" and len(w) == 0:
|
||||
n["widgets_values"] = [2] # keyframes; renderers interpolate frames
|
||||
elif t == "PointcloudTrajectoryEnricher" and len(w) == 20:
|
||||
# old tail: reproject-back proj/fov/w/h + legacy backend/voxel params
|
||||
n["widgets_values"] = w[:13] + [0.07, 3, DA_MODEL]
|
||||
else:
|
||||
continue
|
||||
changed.append(f"{t}#{n['id']}")
|
||||
return changed
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Per-file specs
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def do_demo_camera(name="demo_camera_workflow.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 1, "Rotate camera (pitch 60°)")
|
||||
set_title(d, 2, "Pinhole 90° → Equirect 180°")
|
||||
set_title(d, 7, "Reprojected image")
|
||||
set_title(d, 4, "Coverage mask (white = hole)")
|
||||
add_note(d, (
|
||||
"# Camera reprojection demo\n\n"
|
||||
"Minimal example of the two core nodes: **TransformToMatrix** rotates the "
|
||||
"virtual camera (theta = 60° pitch) and **ReprojectImage** converts a 90° "
|
||||
"pinhole image into a 180° equirectangular view.\n\n"
|
||||
"Previews show the reprojected image and the coverage mask "
|
||||
"(white = pixels the source image cannot see).\n\n"
|
||||
"- **Set:** the image in LoadImage.\n"
|
||||
"- **Requires:** nothing beyond this pack (no models).\n"
|
||||
"- **Try:** switch `output_projection` to FISHEYE, or raise `feathering` "
|
||||
"to soften the mask edge."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_outpaint_node_test(name="Outpaint_node_test.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 11, "Outpaint one patch (yaw +45°)")
|
||||
set_title(d, 3, "Result")
|
||||
set_title(d, 5, "Holes still to fill")
|
||||
add_note(d, (
|
||||
"# OutpaintAnyProjection smoke test\n\n"
|
||||
"Single-node sanity check: a 90° pinhole image is placed on a 180° fisheye "
|
||||
"canvas and one 90° pinhole patch at yaw +45° is Flux-inpainted "
|
||||
"(10 steps for speed).\n\n"
|
||||
"Outputs: the partially outpainted canvas and the *remaining holes* mask — "
|
||||
"chain more OutpaintAnyProjection nodes at other angles to fill it "
|
||||
"(see `Outpaint_fisheye180.json`).\n\n"
|
||||
"- **Set:** input image; prompt inside the node (empty = unconditional).\n"
|
||||
"- **Requires:** `custom_nodes/inpainting_flux` "
|
||||
"(installed automatically by this pack's install.py) — downloads "
|
||||
"FLUX.1-Fill NF4 weights on first run."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_fisheye_to_pointcloud(name="fisheye_to_pointcloud.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 155, "Metric depth (fisheye-aware)")
|
||||
set_title(d, 160, "Depth → point cloud")
|
||||
set_title(d, 161, "Save .ply / .npy")
|
||||
set_groups(d, [
|
||||
("1. Fisheye metric depth", [154, 155, 156, 157, 158, 159]),
|
||||
("2. Unproject & save", [160, 161]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# Fisheye 180° → point cloud\n\n"
|
||||
"Estimates metric depth directly on a 180° fisheye image — "
|
||||
"**FisheyeDepthEstimator** internally splits it into pinhole views, runs "
|
||||
"Depth-Anything V2 on each and merges the depths back (SOFTMERGE) — then "
|
||||
"unprojects image + depth to a 3D point cloud and saves it.\n\n"
|
||||
"Previews: colorized depth and the validity mask.\n\n"
|
||||
"- **Set:** fisheye input image (e.g. produced by "
|
||||
"`Outpaint_fisheye180.json`); filename in SavePointCloud.\n"
|
||||
"- **Requires:** Depth-Anything V2 (auto-downloads from HuggingFace).\n"
|
||||
"- **Next:** view the cloud with `test_pointcloud_loading.json` or "
|
||||
"synthesize a second eye with `sbs180_workflow.json`."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_pointcloud(name="PointCloud.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 6, "Move camera (dolly −0.1, yaw 20°)")
|
||||
set_title(d, 31, "Render novel view")
|
||||
set_title(d, 47, "Drop stretched/occluded points")
|
||||
set_title(d, 43, "Novel-view depth → image")
|
||||
set_groups(d, [
|
||||
("1. Image → metric ray depth", [1, 38, 45, 42, 40]),
|
||||
("2. Lift to 3D & move camera", [18, 6, 9, 47]),
|
||||
("3. Novel view & previews", [31, 13, 12, 43, 44, 25, 26]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# Single image → point cloud → novel view\n\n"
|
||||
"Lifts one pinhole image to a 3D point cloud via monocular metric depth, "
|
||||
"moves the camera, cleans occlusion artifacts and re-renders from the new "
|
||||
"viewpoint.\n\n"
|
||||
"Stages: DepthEstimator → ZDepthToRayDepth → DepthToPointCloud → "
|
||||
"TransformPointCloud (dolly −0.1, yaw 20°) → PointCloudCleaner → "
|
||||
"ProjectPointCloud. MedianFilterImage smooths the re-render "
|
||||
"(from ComfyUI-Image-Filters, optional).\n\n"
|
||||
"- **Set:** input image; the camera move in TransformToMatrix.\n"
|
||||
"- **Requires:** Depth-Anything V2 (auto-download); ComfyUI-Image-Filters "
|
||||
"only for the median-filter preview.\n"
|
||||
"- **Note:** PointCloudCleaner params were reset to defaults during the "
|
||||
"2026-07 schema migration — retune voxel_size / min_points_per_voxel if "
|
||||
"the render looks too sparse."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_pointcloud_walker(name="Pointcloud walker.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 5, "Camera move (edit me)")
|
||||
set_title(d, 17, "Build trajectory")
|
||||
set_title(d, 14, "Render fly-through")
|
||||
set_groups(d, [
|
||||
("1. Image → point cloud", [1, 3, 20, 2]),
|
||||
("2. Trajectory", [5, 4, 17]),
|
||||
("3. Render & save", [14, 10]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# Point cloud walker (fly-through video)\n\n"
|
||||
"Single image → metric depth → point cloud, then **CameraTrajectoryNode** "
|
||||
"derives a camera path and **CameraMotionNode** renders a fly-through "
|
||||
"saved as WEBM.\n\n"
|
||||
"- **Set:** input image; the move in TransformToMatrix "
|
||||
"(shift XYZ + theta/phi); frames-per-segment (`n_points`) in "
|
||||
"CameraMotionNode.\n"
|
||||
"- **Requires:** Depth-Anything V2 (auto-download).\n"
|
||||
"- **Note:** the pinhole FOV here is 60° — keep DepthToPointCloud and "
|
||||
"CameraMotionNode FOVs consistent with your input."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_test_pointcloud_loading(name="Test pointcloud_loading.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 6, "Load saved cloud (.ply/.npy)")
|
||||
set_title(d, 1, "Start pose (identity)")
|
||||
set_title(d, 5, "End pose (dolly +0.1)")
|
||||
set_title(d, 7, "Render motion")
|
||||
add_note(d, (
|
||||
"# Load a saved point cloud & orbit\n\n"
|
||||
"Reloads a point cloud saved by SavePointCloud and renders a short camera "
|
||||
"move between two poses (identity → 0.1 forward) as a WEBM.\n\n"
|
||||
"- **Set:** the file in LoadPointCloud (dropdown lists the ComfyUI input "
|
||||
"dir — run `PointCloud.json` or `fisheye_to_pointcloud.json` first); the "
|
||||
"two TransformToMatrix poses.\n"
|
||||
"- **Requires:** nothing beyond this pack."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_pc_enricher(name="PC_enricher.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 3, "Enrich cloud along trajectory (Flux outpaint)")
|
||||
set_title(d, 11, "Orbit preview of enriched cloud")
|
||||
set_groups(d, [
|
||||
("1. Fisheye → point cloud", [13, 14, 17, 16, 15]),
|
||||
("2. Enrich along trajectory", [18, 3]),
|
||||
("3. Save & preview", [4, 11, 12]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# Point cloud enricher (outpaint along a trajectory)\n\n"
|
||||
"Fisheye image → metric depth → point cloud, then "
|
||||
"**PointcloudTrajectoryEnricher** walks the loaded camera trajectory: at "
|
||||
"each pose it renders the cloud, Flux-inpaints the disocclusion holes, "
|
||||
"re-estimates depth, aligns it and merges the new points into the cloud. "
|
||||
"The enriched cloud is saved and previewed as an orbit video.\n\n"
|
||||
"This is the one-node version of `pointcloud_inpaint.json`.\n\n"
|
||||
"- **Set:** input image; trajectory .npy (record one with "
|
||||
"SaveTrajectory); prompt inside the enricher.\n"
|
||||
"- **Requires:** inpainting_flux (auto-installed), Depth-Anything V2, "
|
||||
"FLUX.1-Fill NF4 weights.\n"
|
||||
"- **Note:** 2026-07 schema migration — the enricher's render/reproject "
|
||||
"options are now internal; voxel merge params were reset to defaults."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_fisheye_depth(name="Fisheye_depth_workflow.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
retitle_groups(d, [
|
||||
"View 1: center pinhole 90°",
|
||||
"View 2: yaw +45°",
|
||||
"View 3: yaw −45°",
|
||||
"View 4: pitch +45°",
|
||||
"View 5: pitch −45°",
|
||||
"View 6: full-fisheye fallback",
|
||||
])
|
||||
# fusion chain + export were never grouped
|
||||
d["groups"].append({
|
||||
"id": 7,
|
||||
"title": "Fuse views & export point cloud",
|
||||
"bounding": bbox(d, [95, 146, 147, 148, 149, 105, 151, 153, 154]),
|
||||
"color": "#b58b2a",
|
||||
"font_size": 24,
|
||||
"flags": {},
|
||||
})
|
||||
set_title(d, 149, "Final fuse (SRC keeps fused views)")
|
||||
set_title(d, 151, "Fused depth → point cloud")
|
||||
add_note(d, (
|
||||
"# Multi-view fisheye depth — manual reference\n\n"
|
||||
"Estimates consistent metric depth over a 180° fisheye image by "
|
||||
"reprojecting it into five pinhole views (center, yaw ±45°, pitch ±45°), "
|
||||
"running Depth-Anything V2 on each, reprojecting the depths back to the "
|
||||
"fisheye and progressively fusing them (CombineDepthsNode + "
|
||||
"DepthRenormalizer to align scales), plus a full-fisheye pass for the "
|
||||
"rim. The fused depth is unprojected and saved as a point cloud.\n\n"
|
||||
"⚠️ **This whole graph is now one node** — `FisheyeDepthEstimator` does "
|
||||
"the same split-and-merge internally (see `fisheye_to_pointcloud.json`). "
|
||||
"Kept as a transparent, tweakable reference implementation.\n\n"
|
||||
"- **Set:** fisheye input image; SavePointCloud filename.\n"
|
||||
"- **Requires:** Depth-Anything V2 (auto-download)."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_outpaint_fisheye180(name="Outpaint_fisheye180.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
retitle_groups(d, [
|
||||
"2. Chained patch outpaints",
|
||||
"1. Pinhole → fisheye canvas",
|
||||
])
|
||||
set_title(d, 2, "Outpaint yaw +45°")
|
||||
set_title(d, 3, "Outpaint yaw −45°")
|
||||
set_title(d, 4, "Outpaint pitch +45°")
|
||||
set_title(d, 5, "Outpaint pitch −45°")
|
||||
set_title(d, 9, "Full-frame pass (low-res rim fill)")
|
||||
set_title(d, 15, "Upscale rim fill to 4096")
|
||||
set_title(d, 12, "Composite sharp patches over rim")
|
||||
add_note(d, (
|
||||
"# Outpaint to a full 180° fisheye\n\n"
|
||||
"Places a 90° pinhole image onto a 180° fisheye canvas (ReprojectImage), "
|
||||
"then chains **five OutpaintAnyProjection passes** — yaw +45°, yaw −45°, "
|
||||
"pitch +45°, pitch −45°, and a low-res full-frame pass for the rim. Each "
|
||||
"pass consumes the previous pass's *remaining holes* mask. A PorterDuff "
|
||||
"composite keeps the sharp high-res patches on top of the upscaled rim "
|
||||
"fill.\n\n"
|
||||
"- **Set:** input image; prompts inside each outpaint node (optional).\n"
|
||||
"- **Requires:** inpainting_flux (auto-installed by install.py); "
|
||||
"FLUX.1-Fill NF4 weights download on first run (~12 GB VRAM).\n"
|
||||
"- **Output:** `Saved_fisheye` PNG — used as the input of "
|
||||
"`fisheye_to_pointcloud.json`, `sbs180_workflow.json` and "
|
||||
"`PC_enricher.json`."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_outpainting_fisheye_sd(name="outpainting_fisheye.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
# --- structural repair: pre-2025 ReprojectImage stored patch rotations as
|
||||
# widgets; the current node takes a MAT_4X4 transform_matrix input and an
|
||||
# inverse flag instead (mirrors outpainting_fisheye_flux.json).
|
||||
fixes = {
|
||||
42: ([90, 180, "PINHOLE", "FISHEYE", 4096, 4096, False, 0], None),
|
||||
40: ([180, 90, "FISHEYE", "PINHOLE", 1024, 1024, False, 0], "+42"),
|
||||
51: ([90, 180, "PINHOLE", "FISHEYE", 4096, 4096, True, 0], "+42"),
|
||||
59: ([180, 90, "FISHEYE", "PINHOLE", 1024, 1024, False, 0], "-42"),
|
||||
63: ([90, 180, "PINHOLE", "FISHEYE", 4096, 4096, True, 0], "-42"),
|
||||
67: ([180, 180, "FISHEYE", "FISHEYE", 1024, 1024, False, 0], None),
|
||||
72: ([180, 180, "FISHEYE", "FISHEYE", 4096, 4096, False, 0], None),
|
||||
}
|
||||
needs_repair = any(len(node(d, nid).get("widgets_values", [])) == 9
|
||||
for nid in fixes)
|
||||
if needs_repair:
|
||||
m_pos = add_transform_node(d, [20, 480], [0, 0, 0, 0, 42])
|
||||
m_neg = add_transform_node(d, [20, 1180], [0, 0, 0, 0, -42])
|
||||
set_title(d, m_pos, "Patch rotation +42°")
|
||||
set_title(d, m_neg, "Patch rotation −42°")
|
||||
for nid, (widgets, mat) in fixes.items():
|
||||
node(d, nid)["widgets_values"] = widgets
|
||||
if mat is not None:
|
||||
link(d, m_pos if mat == "+42" else m_neg, nid, "transform_matrix")
|
||||
retitle_groups(d, [
|
||||
"Patch 1: yaw +42° — extract & inpaint-encode",
|
||||
"Patch 2: yaw −42° — extract & inpaint-encode",
|
||||
])
|
||||
set_title(d, 42, "Pinhole 90° → fisheye canvas")
|
||||
set_title(d, 67, "Full-frame pass (low-res)")
|
||||
add_note(d, (
|
||||
"# Outpaint fisheye — SD-inpaint-checkpoint variant\n\n"
|
||||
"Same pipeline as `outpainting_fisheye_flux.json`, but the holes are "
|
||||
"filled with a classic SD inpainting checkpoint (VAEEncodeForInpaint + "
|
||||
"KSampler) instead of Flux: pinhole 90° → 180° fisheye canvas, two ±42° "
|
||||
"pinhole patches and a final full-frame pass are inpainted and "
|
||||
"composited; RealESRGAN upscales the result.\n\n"
|
||||
"- **Set:** input image; positive/negative prompts.\n"
|
||||
"- **Requires (models):** `512-inpainting-ema.safetensors`, "
|
||||
"`RealESRGAN_x4plus.pth`. No custom node packs.\n"
|
||||
"- **Repaired 2026-07:** patch rotations were stored in a pre-2025 "
|
||||
"ReprojectImage schema; they are now explicit TransformToMatrix (±42°) "
|
||||
"nodes + `inverse` flags. Prefer the Flux variant for quality."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_outpainting_fisheye_flux(name="outpainting_fisheye_flux.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
for nid, inverse in ((42, False), (40, False), (51, True), (80, False),
|
||||
(63, True), (67, False), (72, False)):
|
||||
w = node(d, nid)["widgets_values"]
|
||||
if len(w) == 8:
|
||||
w[6] = inverse # was stored as 0/45/true mixtures
|
||||
retitle_groups(d, [
|
||||
"Patch 1: yaw +42° — extract & Flux inpaint",
|
||||
"Patch 2: yaw −42° — extract & Flux inpaint",
|
||||
])
|
||||
set_title(d, 42, "Pinhole 90° → fisheye canvas")
|
||||
set_title(d, 75, "Patch rotation +42°")
|
||||
set_title(d, 76, "Patch rotation −42°")
|
||||
set_title(d, 67, "Full-frame pass (low-res)")
|
||||
set_title(d, 77, "Upscale full pass to 4096")
|
||||
add_note(d, (
|
||||
"# Outpaint fisheye — Flux variant\n\n"
|
||||
"Pinhole 90° image → 180° fisheye canvas; **Flux Inpainting** fills two "
|
||||
"90° pinhole patches (rotations ±42° from the TransformToMatrix nodes) "
|
||||
"and one low-res full-frame fisheye pass; PorterDuff composites + "
|
||||
"RealESRGAN upscale assemble the final 4096² fisheye "
|
||||
"(saved as `fluxfish`).\n\n"
|
||||
"This is the manual, step-visible version of what "
|
||||
"**OutpaintAnyProjection** does in one node — see "
|
||||
"`Outpaint_fisheye180.json`.\n\n"
|
||||
"- **Set:** input image; prompts in the three Flux Inpainting nodes.\n"
|
||||
"- **Requires:** inpainting_flux (auto-installed), FLUX.1-Fill NF4 "
|
||||
"weights (first-run download), `RealESRGAN_x4plus.pth`."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_sbs180(name="sbs180_workflow.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
retitle_groups(d, [
|
||||
"1. Fisheye metric depth",
|
||||
"2. Point cloud & eye-baseline shift",
|
||||
"3. Outpaint disocclusions & export equirect",
|
||||
"Previews",
|
||||
])
|
||||
set_title(d, 33, "Eye baseline (shiftX 0.1)")
|
||||
set_title(d, 37, "Right eye (equirect)")
|
||||
set_title(d, 44, "Left eye (equirect)")
|
||||
add_note(d, (
|
||||
"# SBS VR180: synthesize the second eye\n\n"
|
||||
"From one 180° fisheye view, synthesizes a stereo pair: fisheye metric "
|
||||
"depth → point cloud → clean → shift the camera by the eye baseline "
|
||||
"(TransformToMatrix shiftX = 0.1) → re-project to fisheye → four chained "
|
||||
"OutpaintAnyProjection passes fill the disocclusions → both eyes are "
|
||||
"exported as 180° equirectangular images.\n\n"
|
||||
"- **Set:** fisheye input (e.g. from `Outpaint_fisheye180.json`); the "
|
||||
"baseline (0.1 ≈ 6.5 cm when depth is metric); prompts optional.\n"
|
||||
"- **Requires:** inpainting_flux (auto-installed), Depth-Anything V2.\n"
|
||||
"- **Output:** `init_camera_equirect` + `shifted_camera_equirect` — "
|
||||
"combine side-by-side for a VR180 player."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_pointcloud_inpaint(name="pointcloud_inpaint.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_title(d, 6, "Camera shift (dolly −0.1)")
|
||||
set_title(d, 43, "Flux inpaint holes")
|
||||
set_title(d, 54, "Align new depth to cloud")
|
||||
set_title(d, 55, "Merge old + new points")
|
||||
set_groups(d, [
|
||||
("1. Image → point cloud", [1, 46, 18]),
|
||||
("2. Novel view & hole mask", [6, 9, 10, 45, 27, 35, 48, 47]),
|
||||
("3. Flux inpaint", [43, 26]),
|
||||
("4. Lift inpainted region & merge", [49, 54, 50, 55]),
|
||||
("5. Orbit render", [61, 67, 68, 69, 63]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# Iterative point-cloud inpainting\n\n"
|
||||
"Enriches a single-image point cloud with generated content: move the "
|
||||
"camera back → render the cloud (holes appear) → grow + invert the "
|
||||
"coverage mask → **Flux-inpaint the holes** → re-estimate depth on the "
|
||||
"inpainted image → **DepthRenormalizer** aligns it to the original "
|
||||
"cloud's depth → lift the new pixels to 3D → **PointCloudUnion** merges "
|
||||
"everything → orbit render of the enriched scene.\n\n"
|
||||
"One-node alternative: PointcloudTrajectoryEnricher "
|
||||
"(`PC_enricher.json`).\n\n"
|
||||
"- **Set:** input image; camera shift; inpaint prompt.\n"
|
||||
"- **Requires:** inpainting_flux (auto-installed), Depth-Anything V2."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_video_camera(name="video_camera.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
set_groups(d, [
|
||||
("1. Load & pad video", [9, 23, 26, 27, 21, 68]),
|
||||
("2. Metric video depth", [19, 12]),
|
||||
("3. Re-render with new camera", [16, 11, 10, 6]),
|
||||
("4. Masks & composite", [39, 41, 60, 75, 54, 74]),
|
||||
("5. Auto-caption (Florence2)", [63, 62, 61]),
|
||||
("6. WAN VACE re-generation", [29, 30, 37, 32, 33, 34, 28, 31, 36, 35]),
|
||||
("7. Outputs", [4, 13, 14, 38]),
|
||||
])
|
||||
set_title(d, 11, "Final camera pose (edit me)")
|
||||
set_title(d, 6, "Re-render along trajectory")
|
||||
set_title(d, 28, "WAN VACE control")
|
||||
add_note(d, (
|
||||
"# Re-shoot a video with a new camera move\n\n"
|
||||
"Re-renders an input video along a user-defined camera trajectory and "
|
||||
"uses WAN 2.1 VACE to regenerate what the new camera reveals:\n\n"
|
||||
"1. Video is padded square and depth-estimated per frame "
|
||||
"(Video-Depth-Anything metric).\n"
|
||||
"2. **VideoCameraMotionSequence** lifts each frame to a point cloud and "
|
||||
"re-renders it along the SE(3)-interpolated trajectory.\n"
|
||||
"3. Disocclusion masks + the re-rendered frames become VACE "
|
||||
"control video/masks; **Florence2** auto-captions the clip as the "
|
||||
"prompt; WAN 2.1 VACE 14B fills the gaps.\n\n"
|
||||
"- **Set:** video path (VHS Load Video Path); the camera move "
|
||||
"(two TransformToMatrix poses); override the auto-caption in "
|
||||
"CLIPTextEncode if desired.\n"
|
||||
"- **Requires (packs):** VideoHelperSuite, ComfyUI-Florence2, KJNodes "
|
||||
"(GetImageRangeFromBatch), ComfyUI-Easy-Use (math), ComfyUI-Image-"
|
||||
"Filters (BlurMaskFast).\n"
|
||||
"- **Requires (models):** `metric_video_depth_anything_vitl.pth` "
|
||||
"(install.sh `depth`), WAN 2.1 VACE 14B + umt5-xxl + WAN VAE "
|
||||
"(install.sh `vae`), Florence-2 (auto-download).\n"
|
||||
"- **Outputs:** re-rendered composite WEBM, depth WEBM, final VACE clip."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def do_wan_vace_ref(name="wan_vace_ref_to_video.json"):
|
||||
d = load(name)
|
||||
repair_widgets(d)
|
||||
# broken model filename: comma + missing underscore
|
||||
n37 = node(d, 37)
|
||||
if n37["widgets_values"][0] == "wan2,1_vace14B_fp16.safetensors":
|
||||
n37["widgets_values"][0] = "wan2.1_vace_14B_fp16.safetensors"
|
||||
set_title(d, 70, "Render control frames + masks")
|
||||
set_title(d, 55, "WAN VACE control")
|
||||
set_groups(d, [
|
||||
("1. Fisheye image → point cloud", [52, 62, 83, 82, 81, 63, 84]),
|
||||
("2. Camera-motion control video", [77, 70, 78, 80]),
|
||||
("3. WAN VACE generation", [37, 54, 38, 6, 7, 39, 55, 3, 56, 8]),
|
||||
("4. Save", [60, 85, 58]),
|
||||
])
|
||||
add_note(d, (
|
||||
"# WAN VACE: still image + camera move → video\n\n"
|
||||
"Turns a single 180° fisheye still into a camera-move video: metric "
|
||||
"depth → point cloud → **CameraMotionNode** renders point-splat frames "
|
||||
"and masks along a loaded trajectory; these become the VACE control "
|
||||
"video/masks with the original image as the reference, and WAN 2.1 VACE "
|
||||
"14B synthesizes the final clip from your prompt.\n\n"
|
||||
"- **Set:** fisheye image; trajectory file (record one with "
|
||||
"SaveTrajectory); the positive prompt.\n"
|
||||
"- **Requires:** WAN 2.1 VACE models (install.sh `vae`), Depth-Anything "
|
||||
"V2, VideoHelperSuite (only for the h264/MP4 export — SaveWEBM works "
|
||||
"without it).\n"
|
||||
"- **Fixed 2026-07:** the UNET filename contained a typo "
|
||||
"(`wan2,1_vace14B` → `wan2.1_vace_14B_fp16.safetensors`)."
|
||||
))
|
||||
save(name, d)
|
||||
|
||||
|
||||
def main():
|
||||
for fn in (
|
||||
do_demo_camera, do_outpaint_node_test, do_fisheye_to_pointcloud,
|
||||
do_pointcloud, do_pointcloud_walker, do_test_pointcloud_loading,
|
||||
do_pc_enricher, do_fisheye_depth, do_outpaint_fisheye180,
|
||||
do_outpainting_fisheye_sd, do_outpainting_fisheye_flux, do_sbs180,
|
||||
do_pointcloud_inpaint, do_video_camera, do_wan_vace_ref,
|
||||
):
|
||||
fn()
|
||||
print(f"[ok] {fn.__name__}")
|
||||
# video_to_4d_world / video_to_4d_walkable_world already have notes+groups;
|
||||
# normalize formatting only.
|
||||
for name in ("video_to_4d_world.json", "video_to_4d_walkable_world.json"):
|
||||
save(name, load(name))
|
||||
print(f"[ok] reformat {name}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,543 @@
|
||||
"""Standalone CPU smoke test for the 4D-world node stack (no ComfyUI, no CUDA,
|
||||
no model downloads).
|
||||
|
||||
Run with:
|
||||
python notebooks/smoke_test_4d.py
|
||||
|
||||
Stubs `folder_paths` via sys.modules injection so the repo modules import
|
||||
outside the ComfyUI runtime, then functionally exercises the NEW code paths
|
||||
with small synthetic data:
|
||||
|
||||
1. interpolate_se3 (pointcloud_nodes, contract C1)
|
||||
2. render_gaussians (GS_nodes, contract C2) shapes + empty case
|
||||
3. render_gaussians fast anisotropic footprint
|
||||
4. GaussianSplats4D.at_time (GS4D_nodes, contract C3)
|
||||
5. BuildSplats4D kNN track binding
|
||||
6. SplitSplatsByMask
|
||||
7. MotionMaskFromDepth
|
||||
8. align_depth_scale (world_nodes, contract C4) + DepthEdgeFilter
|
||||
9. FuseSplats weighted voxel fusion
|
||||
10. SphereSplatSeed pano -> splat sphere -> render round-trip
|
||||
11. Fisheye image-circle exclusion (motion mask / tracks / splat split)
|
||||
"""
|
||||
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
import traceback
|
||||
import types
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Environment setup: repo on sys.path + folder_paths stub (before repo imports)
|
||||
# --------------------------------------------------------------------------- #
|
||||
REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
if REPO_ROOT not in sys.path:
|
||||
sys.path.insert(0, REPO_ROOT)
|
||||
|
||||
_TMP_DIR = tempfile.mkdtemp(prefix="smoke_test_4d_")
|
||||
|
||||
|
||||
def _stub_get_save_image_path(filename_prefix, output_dir, *args, **kwargs):
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
return output_dir, filename_prefix, 0, "", filename_prefix
|
||||
|
||||
|
||||
_fp_stub = types.ModuleType("folder_paths")
|
||||
_fp_stub.get_input_directory = lambda: _TMP_DIR
|
||||
_fp_stub.get_output_directory = lambda: _TMP_DIR
|
||||
_fp_stub.get_temp_directory = lambda: _TMP_DIR
|
||||
_fp_stub.get_save_image_path = _stub_get_save_image_path
|
||||
_fp_stub.get_annotated_filepath = lambda name: os.path.join(_TMP_DIR, name)
|
||||
_fp_stub.exists_annotated_filepath = lambda name: os.path.exists(os.path.join(_TMP_DIR, name))
|
||||
_fp_stub.get_filename_list = lambda folder: []
|
||||
_fp_stub.models_dir = _TMP_DIR
|
||||
sys.modules["folder_paths"] = _fp_stub
|
||||
|
||||
import numpy as np # noqa: E402
|
||||
import torch # noqa: E402
|
||||
|
||||
import GS_nodes # noqa: E402
|
||||
import GS4D_nodes # noqa: E402
|
||||
import pointcloud_nodes # noqa: E402
|
||||
import world_nodes # noqa: E402
|
||||
|
||||
GaussianSplats = GS_nodes.GaussianSplats
|
||||
|
||||
torch.manual_seed(0)
|
||||
np.random.seed(0)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Helpers
|
||||
# --------------------------------------------------------------------------- #
|
||||
def make_splats(
|
||||
xyz: torch.Tensor,
|
||||
sigma: float = 0.05,
|
||||
color: tuple = None,
|
||||
opacity_logit: float = 4.0,
|
||||
) -> GaussianSplats:
|
||||
"""Isotropic sh_order-0 splats at the given positions."""
|
||||
n = xyz.shape[0]
|
||||
if color is None:
|
||||
rgb = torch.rand(n, 3)
|
||||
else:
|
||||
rgb = torch.tensor(color, dtype=torch.float32).view(1, 3).expand(n, 3)
|
||||
C0 = 0.28209479177387814
|
||||
return GaussianSplats(
|
||||
xyz=xyz.float(),
|
||||
scale=torch.full((n, 3), math.log(sigma)),
|
||||
rotation=torch.tensor([1.0, 0.0, 0.0, 0.0]).view(1, 4).expand(n, 4).contiguous(),
|
||||
opacity=torch.full((n, 1), float(opacity_logit)),
|
||||
f_dc=((rgb - 0.5) / C0).contiguous(),
|
||||
f_rest=torch.zeros(n, 0),
|
||||
sh_order=0,
|
||||
)
|
||||
|
||||
|
||||
def rot_x(deg: float) -> torch.Tensor:
|
||||
a = math.radians(deg)
|
||||
return torch.tensor(
|
||||
[[1, 0, 0], [0, math.cos(a), -math.sin(a)], [0, math.sin(a), math.cos(a)]],
|
||||
dtype=torch.float32,
|
||||
)
|
||||
|
||||
|
||||
def rot_y(deg: float) -> torch.Tensor:
|
||||
a = math.radians(deg)
|
||||
return torch.tensor(
|
||||
[[math.cos(a), 0, math.sin(a)], [0, 1, 0], [-math.sin(a), 0, math.cos(a)]],
|
||||
dtype=torch.float32,
|
||||
)
|
||||
|
||||
|
||||
def make_pose(R: torch.Tensor, t) -> torch.Tensor:
|
||||
M = torch.eye(4)
|
||||
M[:3, :3] = R
|
||||
M[:3, 3] = torch.tensor(t, dtype=torch.float32)
|
||||
return M
|
||||
|
||||
|
||||
IDENTITY_4X4 = torch.eye(4)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Tests
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_01_interpolate_se3():
|
||||
poses = torch.stack(
|
||||
[
|
||||
make_pose(torch.eye(3), [0.0, 0.0, 0.0]),
|
||||
make_pose(rot_y(90.0), [1.0, 2.0, 3.0]),
|
||||
make_pose(rot_y(90.0) @ rot_x(45.0), [-1.0, 0.0, 2.0]),
|
||||
]
|
||||
)
|
||||
out = pointcloud_nodes.interpolate_se3(poses, 10)
|
||||
assert out.shape == (10, 4, 4), f"shape {tuple(out.shape)}"
|
||||
|
||||
eye = torch.eye(3)
|
||||
for i in range(10):
|
||||
R = out[i, :3, :3]
|
||||
ortho_err = (R @ R.T - eye).abs().max().item()
|
||||
det = torch.det(R).item()
|
||||
assert ortho_err < 1e-4, f"step {i}: R@R.T deviates from I by {ortho_err}"
|
||||
assert abs(det - 1.0) < 1e-4, f"step {i}: det(R)={det}"
|
||||
assert torch.allclose(out[i, 3], torch.tensor([0.0, 0.0, 0.0, 1.0]), atol=1e-6)
|
||||
|
||||
assert (out[0] - poses[0]).abs().max().item() < 1e-4, "start pose mismatch"
|
||||
assert (out[-1] - poses[-1]).abs().max().item() < 1e-4, "end pose mismatch"
|
||||
|
||||
# K == 1 repeats.
|
||||
rep = pointcloud_nodes.interpolate_se3(poses[:1], 5)
|
||||
assert rep.shape == (5, 4, 4)
|
||||
assert (rep - poses[0]).abs().max().item() < 1e-6
|
||||
|
||||
|
||||
def test_02_render_gaussians_shapes_and_empty():
|
||||
n, H, W = 200, 48, 64
|
||||
xyz = torch.stack(
|
||||
[
|
||||
torch.rand(n) * 2.0 - 1.0,
|
||||
torch.rand(n) * 2.0 - 1.0,
|
||||
torch.rand(n) * 3.0 + 2.0,
|
||||
],
|
||||
dim=-1,
|
||||
)
|
||||
splats = make_splats(xyz, sigma=0.05)
|
||||
|
||||
for projection, fov in (("PINHOLE", 90.0), ("EQUIRECTANGULAR", 360.0)):
|
||||
image, mask, disparity = GS_nodes.render_gaussians(
|
||||
splats, IDENTITY_4X4, projection, fov, W, H,
|
||||
render_mode="fast", device="cpu",
|
||||
)
|
||||
assert image.shape == (1, H, W, 3), f"{projection} image {tuple(image.shape)}"
|
||||
assert mask.shape == (H, W), f"{projection} mask {tuple(mask.shape)}"
|
||||
assert disparity.shape == (1, H, W, 1), f"{projection} disparity {tuple(disparity.shape)}"
|
||||
assert torch.isfinite(image).all() and torch.isfinite(disparity).all()
|
||||
assert float(mask.min()) >= 0.0 and float(mask.max()) <= 1.0 + 1e-6
|
||||
assert float(mask.sum()) > 0.0, f"{projection}: nothing rendered"
|
||||
|
||||
# Empty case: every splat strictly behind a pinhole camera (known past bug:
|
||||
# early return used to yield only 2 outputs).
|
||||
behind = make_splats(xyz * torch.tensor([1.0, 1.0, -1.0]), sigma=0.05)
|
||||
result = GS_nodes.render_gaussians(
|
||||
behind, IDENTITY_4X4, "PINHOLE", 90.0, W, H,
|
||||
render_mode="fast", device="cpu",
|
||||
)
|
||||
assert isinstance(result, tuple) and len(result) == 3, f"empty render returned {len(result)} outputs"
|
||||
image, mask, disparity = result
|
||||
assert image.shape == (1, H, W, 3)
|
||||
assert mask.shape == (H, W)
|
||||
assert disparity.shape == (1, H, W, 1)
|
||||
assert float(mask.sum()) == 0.0
|
||||
|
||||
|
||||
def test_03_fast_mode_anisotropy():
|
||||
H = W = 128
|
||||
ang = math.radians(45.0) / 2.0
|
||||
splats = GaussianSplats(
|
||||
xyz=torch.tensor([[0.0, 0.0, 3.0]]),
|
||||
scale=torch.log(torch.tensor([[0.5, 0.01, 0.01]])),
|
||||
rotation=torch.tensor([[math.cos(ang), 0.0, 0.0, math.sin(ang)]]), # 45 deg about +z
|
||||
opacity=torch.tensor([[6.0]]),
|
||||
f_dc=torch.zeros(1, 3),
|
||||
f_rest=torch.zeros(1, 0),
|
||||
sh_order=0,
|
||||
)
|
||||
image, mask, disparity = GS_nodes.render_gaussians(
|
||||
splats, IDENTITY_4X4, "PINHOLE", 60.0, W, H,
|
||||
render_mode="fast", max_radius=64, device="cpu",
|
||||
)
|
||||
assert float(mask.sum()) > 0.0, "elongated splat rendered nothing"
|
||||
|
||||
# Alpha-weighted pixel covariance of the footprint.
|
||||
ys, xs = torch.meshgrid(
|
||||
torch.arange(H, dtype=torch.float32), torch.arange(W, dtype=torch.float32),
|
||||
indexing="ij",
|
||||
)
|
||||
w = mask.flatten()
|
||||
wsum = w.sum()
|
||||
mx = (w * xs.flatten()).sum() / wsum
|
||||
my = (w * ys.flatten()).sum() / wsum
|
||||
dx = xs.flatten() - mx
|
||||
dy = ys.flatten() - my
|
||||
cxx = (w * dx * dx).sum() / wsum
|
||||
cyy = (w * dy * dy).sum() / wsum
|
||||
cxy = (w * dx * dy).sum() / wsum
|
||||
cov = torch.tensor([[cxx, cxy], [cxy, cyy]])
|
||||
evals, evecs = torch.linalg.eigh(cov)
|
||||
ratio = float(evals[1] / evals[0].clamp(min=1e-8))
|
||||
assert ratio > 2.0, f"footprint not elongated: eigenvalue ratio {ratio:.2f}"
|
||||
|
||||
# Principal axis should be near 45 degrees (rotation honored).
|
||||
major = evecs[:, 1]
|
||||
angle = math.degrees(math.atan2(float(major[1]), float(major[0]))) % 180.0
|
||||
assert abs(angle - 45.0) < 15.0, f"major axis at {angle:.1f} deg, expected ~45"
|
||||
|
||||
|
||||
def test_04_at_time():
|
||||
T = 5
|
||||
canonical = make_splats(torch.tensor([[0.0, 0.0, 2.0], [0.0, 1.0, 3.0]]))
|
||||
static = make_splats(torch.tensor([[5.0, 5.0, 5.0]]))
|
||||
start = torch.tensor([[0.0, 0.0, 2.0], [0.0, 1.0, 3.0]])
|
||||
end = torch.tensor([[1.0, 0.0, 2.0], [0.0, -1.0, 3.0]])
|
||||
ts = torch.linspace(0.0, 1.0, T)
|
||||
trajectories = torch.stack([start + (end - start) * t for t in ts]) # [5,2,3]
|
||||
|
||||
s4d = GS4D_nodes.GaussianSplats4D(
|
||||
static=static, canonical=canonical, trajectories=trajectories, times=ts,
|
||||
)
|
||||
|
||||
mid = s4d.at_time(0.5)
|
||||
assert len(mid) == 3, f"count {len(mid)} != dynamic+static (3)"
|
||||
# Concat order is [static, dynamic].
|
||||
assert torch.allclose(mid.xyz[0], static.xyz[0], atol=1e-6)
|
||||
expected_mid = 0.5 * (start + end)
|
||||
assert torch.allclose(mid.xyz[1:], expected_mid, atol=1e-5), (
|
||||
f"midpoint mismatch: {mid.xyz[1:]} vs {expected_mid}"
|
||||
)
|
||||
|
||||
lo = s4d.at_time(-1.0)
|
||||
hi = s4d.at_time(2.0)
|
||||
assert torch.allclose(lo.xyz[1:], start, atol=1e-5), "t<range should clamp to first step"
|
||||
assert torch.allclose(hi.xyz[1:], end, atol=1e-5), "t>range should clamp to last step"
|
||||
|
||||
|
||||
def test_05_build_splats4d():
|
||||
T = 5
|
||||
ts = torch.linspace(0.0, 1.0, T)
|
||||
# Two control tracks moving apart along x.
|
||||
track_a = torch.stack([torch.tensor([-1.0 - 2.0 * t, 0.0, 2.0]) for t in ts])
|
||||
track_b = torch.stack([torch.tensor([1.0 + 2.0 * t, 0.0, 2.0]) for t in ts])
|
||||
trajectories3d = torch.stack([track_a, track_b], dim=1) # [T,2,3]
|
||||
|
||||
canonical = make_splats(torch.tensor([[-1.05, 0.0, 2.0], [1.05, 0.0, 2.0]]))
|
||||
node = GS4D_nodes.BuildSplats4D()
|
||||
(s4d,) = node.build_splats4d(
|
||||
canonical=canonical,
|
||||
trajectories3d=trajectories3d,
|
||||
reference_index=0,
|
||||
knn=1,
|
||||
rbf_gamma=0.0,
|
||||
device="cpu",
|
||||
)
|
||||
traj = s4d.trajectories
|
||||
assert traj.shape == (T, 2, 3), f"trajectories shape {tuple(traj.shape)}"
|
||||
# Reference timestep: splats stay at their canonical positions.
|
||||
assert torch.allclose(traj[0], canonical.xyz, atol=1e-5)
|
||||
# Each splat follows its nearest track's displacement direction.
|
||||
disp0 = traj[-1, 0] - traj[0, 0]
|
||||
disp1 = traj[-1, 1] - traj[0, 1]
|
||||
assert disp0[0] < -1.0, f"splat 0 should move -x with track A, moved {disp0.tolist()}"
|
||||
assert disp1[0] > 1.0, f"splat 1 should move +x with track B, moved {disp1.tolist()}"
|
||||
assert torch.allclose(traj[-1, 0], torch.tensor([-3.05, 0.0, 2.0]), atol=1e-4)
|
||||
assert torch.allclose(traj[-1, 1], torch.tensor([3.05, 0.0, 2.0]), atol=1e-4)
|
||||
|
||||
|
||||
def test_06_split_splats_by_mask():
|
||||
H = W = 32
|
||||
mask = torch.zeros(H, W)
|
||||
mask[:, : W // 2] = 1.0 # left half white
|
||||
|
||||
# 10 splats projecting into the left half (x<0), 10 into the right half,
|
||||
# 5 behind the camera.
|
||||
jitter = torch.linspace(-0.1, 0.1, 10)
|
||||
left = torch.stack([torch.full((10,), -0.5) + jitter * 0.1, jitter, torch.full((10,), 2.0)], dim=-1)
|
||||
right = torch.stack([torch.full((10,), 0.5) + jitter * 0.1, jitter, torch.full((10,), 2.0)], dim=-1)
|
||||
behind = torch.stack([jitter[:5], jitter[:5], torch.full((5,), -2.0)], dim=-1)
|
||||
splats = make_splats(torch.cat([left, right, behind], dim=0))
|
||||
|
||||
node = GS4D_nodes.SplitSplatsByMask()
|
||||
inside, outside = node.split_splats(
|
||||
splats=splats,
|
||||
mask=mask,
|
||||
projection="PINHOLE",
|
||||
horizontal_fov=90.0,
|
||||
threshold=0.5,
|
||||
camera_matrix=None,
|
||||
device="cpu",
|
||||
)
|
||||
assert len(inside) == 10, f"inside count {len(inside)} != 10"
|
||||
assert len(outside) == 15, f"outside count {len(outside)} != 15 (10 right + 5 behind)"
|
||||
assert (inside.xyz[:, 0] < 0).all(), "inside splats should be the x<0 group"
|
||||
|
||||
|
||||
def test_07_motion_mask_from_depth():
|
||||
T, H, W = 6, 32, 32
|
||||
depth = torch.full((T, H, W), 5.0)
|
||||
r0, r1 = 8, 16
|
||||
for t in range(T):
|
||||
depth[t, r0:r1, r0:r1] = 3.0 + 0.4 * t # depth-changing square patch
|
||||
|
||||
poses = torch.eye(4).unsqueeze(0).expand(T, 4, 4).contiguous()
|
||||
node = GS4D_nodes.MotionMaskFromDepth()
|
||||
(mask,) = node.motion_mask(
|
||||
depth_seq=depth,
|
||||
trajectory=poses,
|
||||
input_projection="PINHOLE",
|
||||
input_horizontal_fov=90.0,
|
||||
threshold=0.10,
|
||||
frame_gap=2,
|
||||
dilate=0,
|
||||
device="cpu",
|
||||
)
|
||||
assert mask.shape == (T, H, W), f"mask shape {tuple(mask.shape)}"
|
||||
|
||||
patch = mask[:, r0:r1, r0:r1]
|
||||
background = mask.clone()
|
||||
background[:, r0:r1, r0:r1] = 0.0
|
||||
patch_mean = float(patch.mean())
|
||||
bg_sum = float(background.sum())
|
||||
assert patch_mean > 0.9, f"moving square under-detected: mean {patch_mean:.3f}"
|
||||
assert bg_sum == 0.0, f"static plane falsely flagged: {bg_sum} pixels"
|
||||
|
||||
|
||||
def test_08_align_depth_scale_and_depth_edge_filter():
|
||||
H = W = 32
|
||||
new_depth = torch.rand(H, W) * 9.0 + 1.0
|
||||
# ref disparity = 0.5 * new disparity + 0.1 (i.e. ref = 2*new before shift).
|
||||
true_scale, true_shift = 0.5, 0.1
|
||||
ref_depth = 1.0 / (true_scale / new_depth + true_shift)
|
||||
valid = torch.ones(H, W)
|
||||
|
||||
aligned, scale, shift = world_nodes.align_depth_scale(
|
||||
new_depth, ref_depth, valid, mode="scale_shift"
|
||||
)
|
||||
assert abs(scale - true_scale) / true_scale < 0.05, f"scale {scale} vs {true_scale}"
|
||||
assert abs(shift - true_shift) / true_shift < 0.05, f"shift {shift} vs {true_shift}"
|
||||
rel_err = float(((aligned - ref_depth).abs() / ref_depth).max())
|
||||
assert rel_err < 0.01, f"aligned depth off by {rel_err:.4f} (rel)"
|
||||
|
||||
# DepthEdgeFilter: a vertical step edge must be masked out, flat kept.
|
||||
depth = torch.full((H, W), 1.0)
|
||||
depth[:, W // 2 :] = 5.0
|
||||
node = pointcloud_nodes.DepthEdgeFilter()
|
||||
(valid_mask,) = node.filter_edges(depth, relative_threshold=0.05, dilate=1)
|
||||
assert valid_mask.shape == (H, W)
|
||||
edge_cols = valid_mask[:, W // 2 - 1 : W // 2 + 1]
|
||||
assert float(edge_cols.max()) == 0.0, "step-edge pixels not masked out"
|
||||
assert float(valid_mask[:, : W // 2 - 3].min()) == 1.0, "flat left region wrongly masked"
|
||||
assert float(valid_mask[:, W // 2 + 3 :].min()) == 1.0, "flat right region wrongly masked"
|
||||
|
||||
|
||||
def test_09_fuse_splats():
|
||||
n = 20
|
||||
voxel = 0.5
|
||||
base = torch.stack(
|
||||
[
|
||||
torch.arange(n, dtype=torch.float32) * voxel + 0.15,
|
||||
torch.full((n,), 0.15),
|
||||
torch.full((n,), 0.15),
|
||||
],
|
||||
dim=-1,
|
||||
)
|
||||
cloud_a = make_splats(base)
|
||||
cloud_b = make_splats(base + 0.2) # same voxels as A (0.15+0.2 < 0.5)
|
||||
|
||||
node = GS_nodes.FuseSplats()
|
||||
(fused,) = node.fuse_splats(cloud_a, cloud_b, voxel, "smart", 1.0, 1.0, device="cpu")
|
||||
assert len(fused) < len(cloud_a) + len(cloud_b), (
|
||||
f"voxel fuse did not reduce: {len(fused)} vs {len(cloud_a) + len(cloud_b)}"
|
||||
)
|
||||
assert len(fused) == n, f"expected one splat per voxel ({n}), got {len(fused)}"
|
||||
|
||||
# Strong weight_a pulls fused positions onto cloud A.
|
||||
(fused_w,) = node.fuse_splats(cloud_a, cloud_b, voxel, "average", 1000.0, 1.0, device="cpu")
|
||||
assert len(fused_w) == n
|
||||
d_a = torch.cdist(fused_w.xyz, cloud_a.xyz).min(dim=1).values
|
||||
d_b = torch.cdist(fused_w.xyz, cloud_b.xyz).min(dim=1).values
|
||||
assert float(d_a.max()) < 0.01, f"fused positions not near cloud A (max dist {float(d_a.max()):.4f})"
|
||||
assert (d_a < d_b).all(), "weight_a=1000 should pull fused splats toward cloud A"
|
||||
|
||||
|
||||
def test_10_sphere_splat_seed():
|
||||
H, W = 64, 128
|
||||
stride = 2
|
||||
color = (0.2, 0.6, 0.9)
|
||||
pano = torch.tensor(color).view(1, 1, 1, 3).expand(1, H, W, 3).contiguous()
|
||||
|
||||
node = world_nodes.SphereSplatSeed()
|
||||
(splats,) = node.seed_sphere(
|
||||
image=pano,
|
||||
horizontal_fov=360.0,
|
||||
radius=5.0,
|
||||
splat_scale_frac=1.5,
|
||||
stride=stride,
|
||||
device="cpu",
|
||||
)
|
||||
expected = (H // stride) * (W // stride)
|
||||
assert abs(len(splats) - expected) <= max(4, expected // 20), (
|
||||
f"splat count {len(splats)} far from expected ~{expected}"
|
||||
)
|
||||
|
||||
image, mask, disparity = GS_nodes.render_gaussians(
|
||||
splats, IDENTITY_4X4, "PINHOLE", 60.0, 64, 64,
|
||||
render_mode="fast", device="cpu",
|
||||
)
|
||||
assert float(mask.sum()) > 0.0, "pinhole render of the sphere seed is empty"
|
||||
solid = mask > 0.9
|
||||
assert bool(solid.any()), "no confidently covered pixels in the render"
|
||||
rendered = image[0][solid] # [K,3]
|
||||
target = torch.tensor(color)
|
||||
err = (rendered.mean(dim=0) - target).abs().max().item()
|
||||
assert err < 0.05, f"color round-trip failed: rendered mean {rendered.mean(dim=0).tolist()} vs {color}"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Runner
|
||||
# --------------------------------------------------------------------------- #
|
||||
def test_11_fisheye_circle_exclusion():
|
||||
"""Fisheye pixels/points outside the image circle (r > 1) must be ignored:
|
||||
corners never become 'dynamic', corner tracks never lift to 3D, and points
|
||||
at view angles beyond fov/2 are not matched against the mask."""
|
||||
T, H, W = 6, 32, 32
|
||||
fov = 180.0
|
||||
|
||||
# --- MotionMaskFromDepth: wildly flickering corners must stay static ----
|
||||
depth = torch.full((T, H, W), 5.0)
|
||||
depth[:, 14:18, 14:18] = torch.linspace(5.0, 2.0, T).view(T, 1, 1) # real motion
|
||||
u = torch.linspace(-1, 1, W).view(1, W).expand(H, W)
|
||||
v = torch.linspace(-1, 1, H).view(H, 1).expand(H, W)
|
||||
corners = (u * u + v * v) > 1.0 + 1e-6
|
||||
for t in range(T): # garbage depth flicker outside the circle
|
||||
depth[t][corners] = 1.0 if t % 2 == 0 else 10.0
|
||||
identity = torch.eye(4).unsqueeze(0).expand(T, 4, 4).contiguous()
|
||||
(dyn,) = GS4D_nodes.MotionMaskFromDepth().motion_mask(
|
||||
depth_seq=depth, trajectory=identity, input_projection="FISHEYE",
|
||||
input_horizontal_fov=fov, threshold=0.10, frame_gap=2, dilate=0,
|
||||
device="cpu",
|
||||
)
|
||||
assert dyn[:, corners].max().item() == 0.0, "fisheye corners were flagged dynamic"
|
||||
assert dyn[:, 14:18, 14:18].max().item() == 1.0, "real in-circle motion missed"
|
||||
|
||||
# --- TracksToTrajectories: corner track must be dropped -----------------
|
||||
tracks = torch.zeros(T, 2, 2)
|
||||
tracks[:, 0, 0] = W / 2.0 # center track
|
||||
tracks[:, 0, 1] = H / 2.0
|
||||
tracks[:, 1, 0] = 0.0 # corner track (u=v=-1, r=1.414)
|
||||
tracks[:, 1, 1] = 0.0
|
||||
visibility = torch.ones(T, 2)
|
||||
traj3d, track_ok = GS4D_nodes.TracksToTrajectories().tracks_to_trajectories(
|
||||
tracks=tracks, visibility=visibility, depth_seq=depth,
|
||||
input_projection="FISHEYE", input_horizontal_fov=fov,
|
||||
min_visible_frac=0.5, device="cpu",
|
||||
)
|
||||
assert bool(track_ok[0]), "in-circle track unexpectedly invalid"
|
||||
assert not bool(track_ok[1]), "out-of-circle corner track was lifted to 3D"
|
||||
|
||||
# --- SplitSplatsByMask: angle beyond fov/2 lands diagonally inside the
|
||||
# [-1,1] square (r=1.33, u=v~0.94) but must not be matched to the mask ---
|
||||
front = torch.tensor([[0.0, 0.0, 1.0]])
|
||||
theta = math.radians(120.0)
|
||||
behind_diag = torch.tensor([[
|
||||
math.sin(theta) * math.cos(math.radians(45.0)),
|
||||
math.sin(theta) * math.sin(math.radians(45.0)),
|
||||
math.cos(theta),
|
||||
]])
|
||||
splats = make_splats(torch.cat([front, behind_diag], dim=0))
|
||||
inside, outside = GS4D_nodes.SplitSplatsByMask().split_splats(
|
||||
splats=splats, mask=torch.ones(H, W), projection="FISHEYE",
|
||||
horizontal_fov=fov, threshold=0.5, device="cpu",
|
||||
)
|
||||
assert inside.xyz.shape[0] == 1, "expected only the in-fov splat inside"
|
||||
assert outside.xyz.shape[0] == 1, "beyond-fov splat must fall outside"
|
||||
|
||||
|
||||
TESTS = [
|
||||
test_01_interpolate_se3,
|
||||
test_02_render_gaussians_shapes_and_empty,
|
||||
test_03_fast_mode_anisotropy,
|
||||
test_04_at_time,
|
||||
test_05_build_splats4d,
|
||||
test_06_split_splats_by_mask,
|
||||
test_07_motion_mask_from_depth,
|
||||
test_08_align_depth_scale_and_depth_edge_filter,
|
||||
test_09_fuse_splats,
|
||||
test_10_sphere_splat_seed,
|
||||
test_11_fisheye_circle_exclusion,
|
||||
]
|
||||
|
||||
|
||||
def main() -> int:
|
||||
passed = 0
|
||||
failed = []
|
||||
for test in TESTS:
|
||||
name = test.__name__
|
||||
try:
|
||||
test()
|
||||
except Exception:
|
||||
failed.append(name)
|
||||
print(f"[FAIL] {name}")
|
||||
traceback.print_exc()
|
||||
else:
|
||||
passed += 1
|
||||
print(f"[ ok ] {name}")
|
||||
print(f"\n{passed}/{len(TESTS)} tests passed")
|
||||
if failed:
|
||||
print("Failed:", ", ".join(failed))
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,199 @@
|
||||
"""Offline tests for install.py logic and the flux-pack import resolution in
|
||||
flux_fisheye_filling_nodes.py. Stubs pip/git so nothing is actually installed.
|
||||
|
||||
Run: python notebooks/test_install_logic.py
|
||||
"""
|
||||
|
||||
import importlib.util
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
import types
|
||||
|
||||
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
PASS = []
|
||||
|
||||
|
||||
def ok(name):
|
||||
PASS.append(name)
|
||||
print(f"[ ok ] {name}")
|
||||
|
||||
|
||||
def load_install_module():
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"camera_install", os.path.join(REPO, "install.py")
|
||||
)
|
||||
mod = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(mod)
|
||||
return mod
|
||||
|
||||
|
||||
def stub_commands(mod):
|
||||
"""Replace subprocess-based helpers with recorders."""
|
||||
calls = []
|
||||
mod._run = lambda cmd, cwd=None: calls.append((tuple(cmd), cwd))
|
||||
mod._pip_install = lambda *args: calls.append((("pip",) + args, None))
|
||||
mod._git = lambda: "git"
|
||||
return calls
|
||||
|
||||
|
||||
def test_sharp_early_return():
|
||||
mod = load_install_module()
|
||||
calls = stub_commands(mod)
|
||||
# Build the "already materialized" state explicitly — the repo's own
|
||||
# submodule may not be checked out (e.g. CI clones without --recursive).
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
mod.NODE_DIR = os.path.join(tmp, "camera-comfyUI")
|
||||
os.makedirs(os.path.join(mod.NODE_DIR, "submodules", "ml-sharpt", "src", "sharp"))
|
||||
mod.ensure_sharp_checkout()
|
||||
assert calls == [], f"expected no commands, got {calls}"
|
||||
ok("sharp checkout present -> no git calls")
|
||||
|
||||
|
||||
def test_sharp_clone_fallback():
|
||||
mod = load_install_module()
|
||||
calls = stub_commands(mod)
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
mod.NODE_DIR = os.path.join(tmp, "custom_nodes", "camera-comfyUI")
|
||||
os.makedirs(mod.NODE_DIR)
|
||||
# No .git -> should go straight to direct clone
|
||||
mod.ensure_sharp_checkout()
|
||||
assert len(calls) == 1 and "clone" in calls[0][0], calls
|
||||
assert mod.SHARP_GIT_URL in calls[0][0], calls
|
||||
ok("sharp missing + no .git -> direct clone")
|
||||
|
||||
|
||||
def test_vggt_uses_https_git():
|
||||
mod = load_install_module()
|
||||
calls = stub_commands(mod)
|
||||
mod._importable = lambda name: False
|
||||
mod.ensure_vggt()
|
||||
assert calls == [(("pip", "vggt @ git+https://github.com/facebookresearch/vggt.git"), None)], calls
|
||||
ok("vggt -> pip install from git+https (no ssh)")
|
||||
|
||||
|
||||
def test_flux_pack_skip_outside_comfyui():
|
||||
mod = load_install_module()
|
||||
calls = stub_commands(mod)
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
mod.NODE_DIR = os.path.join(tmp, "somewhere", "camera-comfyUI")
|
||||
os.makedirs(mod.NODE_DIR)
|
||||
mod.ensure_flux_inpainting_pack()
|
||||
assert calls == [], calls
|
||||
ok("flux pack: skipped when parent is not custom_nodes")
|
||||
|
||||
|
||||
def test_flux_pack_detects_existing_and_clones_when_missing():
|
||||
mod = load_install_module()
|
||||
for existing in ("inpainting_flux", "ComfyUI-Flux-Inpainting", "ComfyUI-Flux-Inpainting-main"):
|
||||
calls = stub_commands(mod)
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
cn = os.path.join(tmp, "custom_nodes")
|
||||
mod.NODE_DIR = os.path.join(cn, "camera-comfyUI")
|
||||
os.makedirs(mod.NODE_DIR)
|
||||
os.makedirs(os.path.join(cn, existing))
|
||||
mod.ensure_flux_inpainting_pack()
|
||||
assert calls == [], f"{existing}: {calls}"
|
||||
ok("flux pack: all existing folder names detected, no clone")
|
||||
|
||||
calls = stub_commands(mod)
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
cn = os.path.join(tmp, "custom_nodes")
|
||||
mod.NODE_DIR = os.path.join(cn, "camera-comfyUI")
|
||||
os.makedirs(mod.NODE_DIR)
|
||||
mod.ensure_flux_inpainting_pack()
|
||||
assert len(calls) == 1 and "clone" in calls[0][0], calls
|
||||
assert calls[0][0][-1] == os.path.join(cn, "inpainting_flux"), calls
|
||||
ok("flux pack: missing -> cloned as custom_nodes/inpainting_flux")
|
||||
|
||||
|
||||
def test_example_inputs_copied():
|
||||
mod = load_install_module()
|
||||
stub_commands(mod)
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
cn = os.path.join(tmp, "ComfyUI", "custom_nodes")
|
||||
mod.NODE_DIR = os.path.join(cn, "camera-comfyUI")
|
||||
src = os.path.join(mod.NODE_DIR, "example_inputs")
|
||||
os.makedirs(src)
|
||||
with open(os.path.join(src, "example.jpg"), "w") as f:
|
||||
f.write("x")
|
||||
input_dir = os.path.join(tmp, "ComfyUI", "input")
|
||||
os.makedirs(input_dir)
|
||||
with open(os.path.join(input_dir, "existing.jpg"), "w") as f:
|
||||
f.write("keep me")
|
||||
mod.ensure_example_inputs()
|
||||
mod.ensure_example_inputs() # idempotent, no overwrite
|
||||
assert os.path.isfile(os.path.join(input_dir, "example.jpg"))
|
||||
assert open(os.path.join(input_dir, "existing.jpg")).read() == "keep me"
|
||||
ok("example inputs copied to ComfyUI input dir, no overwrite")
|
||||
|
||||
|
||||
def test_main_never_fails():
|
||||
mod = load_install_module()
|
||||
|
||||
def boom():
|
||||
raise RuntimeError("no network")
|
||||
|
||||
mod.STEPS = (("step-a", boom), ("step-b", boom))
|
||||
assert mod.main() == 0
|
||||
ok("main() returns 0 even when every step fails")
|
||||
|
||||
|
||||
def test_flux_import_resolution():
|
||||
"""Build a fake ComfyUI tree and check _import_flux_inpainting finds the
|
||||
pack under a dashed folder name and honors its relative imports."""
|
||||
if "PIL" not in sys.modules:
|
||||
try:
|
||||
import PIL # noqa: F401
|
||||
except ImportError:
|
||||
pil = types.ModuleType("PIL")
|
||||
pil.Image = types.SimpleNamespace()
|
||||
sys.modules["PIL"] = pil
|
||||
sys.modules["PIL.Image"] = types.ModuleType("PIL.Image")
|
||||
|
||||
tmp = tempfile.mkdtemp()
|
||||
try:
|
||||
cn = os.path.join(tmp, "ComfyUI", "custom_nodes")
|
||||
pack = os.path.join(cn, "campack")
|
||||
os.makedirs(pack)
|
||||
for fname in ("reprojection_nodes.py", "flux_fisheye_filling_nodes.py"):
|
||||
shutil.copy(os.path.join(REPO, fname), pack)
|
||||
open(os.path.join(pack, "__init__.py"), "w").close()
|
||||
|
||||
# Fake flux pack under a dashed (non-identifier) folder name with a
|
||||
# relative import, mirroring the real repo layout.
|
||||
flux = os.path.join(cn, "ComfyUI-Flux-Inpainting")
|
||||
os.makedirs(os.path.join(flux, "modules"))
|
||||
open(os.path.join(flux, "__init__.py"), "w").close()
|
||||
open(os.path.join(flux, "modules", "__init__.py"), "w").close()
|
||||
with open(os.path.join(flux, "modules", "load_util.py"), "w") as f:
|
||||
f.write("MARKER = 'loaded'\n")
|
||||
with open(os.path.join(flux, "nodes.py"), "w") as f:
|
||||
f.write(
|
||||
"from .modules.load_util import MARKER\n"
|
||||
"class FluxNF4Inpainting:\n"
|
||||
" marker = MARKER\n"
|
||||
)
|
||||
|
||||
sys.path.insert(0, os.path.dirname(pack))
|
||||
mod = importlib.import_module("campack.flux_fisheye_filling_nodes")
|
||||
assert mod._flux_import_error is None, mod._flux_import_error
|
||||
assert mod.FluxInpainting is not None
|
||||
assert mod.FluxInpainting.marker == "loaded"
|
||||
ok("flux import: dashed folder name resolved incl. relative imports")
|
||||
finally:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
test_sharp_early_return()
|
||||
test_sharp_clone_fallback()
|
||||
test_vggt_uses_https_git()
|
||||
test_flux_pack_skip_outside_comfyui()
|
||||
test_flux_pack_detects_existing_and_clones_when_missing()
|
||||
test_example_inputs_copied()
|
||||
test_main_never_fails()
|
||||
test_flux_import_resolution()
|
||||
print(f"\n{len(PASS)} install-logic checks passed")
|
||||
@@ -0,0 +1,133 @@
|
||||
"""Validate workflows/*.json against the current node definitions.
|
||||
|
||||
For every camera-comfyUI node used in a workflow, compares the stored
|
||||
widgets_values against the widget list derived from the node's INPUT_TYPES
|
||||
(required + optional, in order, counting only widget-type inputs). A length
|
||||
mismatch means the workflow predates a node-schema change and will load with
|
||||
silently shifted/défault values.
|
||||
|
||||
Run: python notebooks/validate_workflows.py
|
||||
Exit code 1 if any mismatch is found (missing node types are also reported).
|
||||
"""
|
||||
|
||||
import glob
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
import types
|
||||
|
||||
REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
if REPO_ROOT not in sys.path:
|
||||
sys.path.insert(0, REPO_ROOT)
|
||||
|
||||
_TMP_DIR = tempfile.mkdtemp(prefix="wf_validate_")
|
||||
|
||||
|
||||
def _stub_get_save_image_path(filename_prefix, output_dir, *args, **kwargs):
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
return output_dir, filename_prefix, 0, "", filename_prefix
|
||||
|
||||
|
||||
_fp = types.ModuleType("folder_paths")
|
||||
_fp.get_input_directory = lambda: _TMP_DIR
|
||||
_fp.get_output_directory = lambda: _TMP_DIR
|
||||
_fp.get_temp_directory = lambda: _TMP_DIR
|
||||
_fp.get_save_image_path = _stub_get_save_image_path
|
||||
_fp.get_annotated_filepath = lambda name: os.path.join(_TMP_DIR, name)
|
||||
_fp.exists_annotated_filepath = lambda name: os.path.exists(os.path.join(_TMP_DIR, name))
|
||||
_fp.get_filename_list = lambda folder: []
|
||||
_fp.models_dir = _TMP_DIR
|
||||
sys.modules["folder_paths"] = _fp
|
||||
|
||||
# Light stubs for heavy deps that some modules import at module level but that
|
||||
# INPUT_TYPES itself does not need.
|
||||
for name in ("transformers", "diffusers"):
|
||||
if name not in sys.modules:
|
||||
try:
|
||||
__import__(name)
|
||||
except ImportError:
|
||||
stub = types.ModuleType(name)
|
||||
stub.pipeline = lambda *a, **k: None
|
||||
sys.modules[name] = stub
|
||||
|
||||
# Import the repo as a package so modules with relative imports load too.
|
||||
import importlib.util # noqa: E402
|
||||
|
||||
_spec = importlib.util.spec_from_file_location(
|
||||
"camcomfy",
|
||||
os.path.join(REPO_ROOT, "__init__.py"),
|
||||
submodule_search_locations=[REPO_ROOT],
|
||||
)
|
||||
_pkg = importlib.util.module_from_spec(_spec)
|
||||
sys.modules["camcomfy"] = _pkg
|
||||
_spec.loader.exec_module(_pkg)
|
||||
NODE_CLASS_MAPPINGS = dict(_pkg.NODE_CLASS_MAPPINGS)
|
||||
|
||||
WIDGET_TYPES = {"INT", "FLOAT", "STRING", "BOOLEAN"}
|
||||
|
||||
|
||||
def widget_specs(cls):
|
||||
"""Ordered (name, combo_options|None) for a node's widget inputs."""
|
||||
it = cls.INPUT_TYPES()
|
||||
specs = []
|
||||
for section in ("required", "optional"):
|
||||
for name, spec in it.get(section, {}).items():
|
||||
t = spec[0] if isinstance(spec, (tuple, list)) and spec else spec
|
||||
if isinstance(t, (list, tuple)): # combo box (either sequence type)
|
||||
specs.append((name, list(t)))
|
||||
elif isinstance(t, str) and t in WIDGET_TYPES:
|
||||
specs.append((name, None))
|
||||
# ComfyUI appends a control_after_generate widget after seeds
|
||||
if t == "INT" and name in ("seed", "noise_seed"):
|
||||
specs.append((f"{name}:control_after_generate", None))
|
||||
return specs
|
||||
|
||||
|
||||
def main() -> int:
|
||||
problems = 0
|
||||
for path in sorted(glob.glob(os.path.join(REPO_ROOT, "workflows", "**", "*.json"),
|
||||
recursive=True)):
|
||||
data = json.load(open(path, encoding="utf-8"))
|
||||
header_shown = False
|
||||
|
||||
def report(msg):
|
||||
nonlocal header_shown, problems
|
||||
if not header_shown:
|
||||
print(f"\n=== {os.path.basename(path)}")
|
||||
header_shown = True
|
||||
print(f" {msg}")
|
||||
problems += 1
|
||||
|
||||
for n in data.get("nodes", []):
|
||||
t = n["type"]
|
||||
if t not in NODE_CLASS_MAPPINGS:
|
||||
continue # builtin or third-party node
|
||||
specs = widget_specs(NODE_CLASS_MAPPINGS[t])
|
||||
got = n.get("widgets_values") or []
|
||||
if isinstance(got, dict):
|
||||
continue # API-style dict widgets (third-party save format)
|
||||
if len(got) != len(specs):
|
||||
report(
|
||||
f"N{n['id']} {t}: {len(got)} widget values, node now has "
|
||||
f"{len(specs)} widgets {[s[0] for s in specs]}; "
|
||||
f"stored={json.dumps(got)[:100]}"
|
||||
)
|
||||
continue
|
||||
for (name, options), value in zip(specs, got):
|
||||
# empty option lists are dynamic file dropdowns (input dir
|
||||
# listing) — not verifiable outside a real ComfyUI install
|
||||
if options and value not in options:
|
||||
report(
|
||||
f"N{n['id']} {t}: widget '{name}' has stale value "
|
||||
f"{value!r}, valid options are {options}"
|
||||
)
|
||||
if problems:
|
||||
print(f"\n{problems} mismatches found")
|
||||
return 1
|
||||
print("all workflow widget schemas match current node definitions")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -7,7 +7,10 @@ import os
|
||||
import folder_paths
|
||||
import logging
|
||||
import hashlib
|
||||
from kornia.filters import median_blur
|
||||
try:
|
||||
from kornia.filters import median_blur
|
||||
except ImportError: # kornia is optional; median_blur is not used in this module
|
||||
median_blur = None
|
||||
|
||||
from tqdm import tqdm
|
||||
# Try importing open3d and its visualization modules; log a warning if not found
|
||||
@@ -136,6 +139,116 @@ def project_first_hit(volume_sparse: torch.Tensor) -> Tuple[torch.Tensor, torch.
|
||||
|
||||
return rgba.permute(2, 0, 1), first_hit.any(dim=2)
|
||||
|
||||
# ==== SE(3) trajectory interpolation ==== #
|
||||
def _rotmat_to_quat_wxyz(R: torch.Tensor) -> torch.Tensor:
|
||||
"""
|
||||
Convert a batch of rotation matrices [K,3,3] to unit quaternions [K,4] (wxyz).
|
||||
Uses Shepperd's method for numerical robustness. K is expected to be small
|
||||
(trajectory waypoints), so a Python loop is acceptable.
|
||||
"""
|
||||
quats = []
|
||||
for i in range(R.shape[0]):
|
||||
m = R[i]
|
||||
trace = m[0, 0] + m[1, 1] + m[2, 2]
|
||||
if trace > 0.0:
|
||||
s = torch.sqrt(trace + 1.0) * 2.0
|
||||
w = 0.25 * s
|
||||
x = (m[2, 1] - m[1, 2]) / s
|
||||
y = (m[0, 2] - m[2, 0]) / s
|
||||
z = (m[1, 0] - m[0, 1]) / s
|
||||
elif m[0, 0] > m[1, 1] and m[0, 0] > m[2, 2]:
|
||||
s = torch.sqrt(1.0 + m[0, 0] - m[1, 1] - m[2, 2]) * 2.0
|
||||
w = (m[2, 1] - m[1, 2]) / s
|
||||
x = 0.25 * s
|
||||
y = (m[0, 1] + m[1, 0]) / s
|
||||
z = (m[0, 2] + m[2, 0]) / s
|
||||
elif m[1, 1] > m[2, 2]:
|
||||
s = torch.sqrt(1.0 + m[1, 1] - m[0, 0] - m[2, 2]) * 2.0
|
||||
w = (m[0, 2] - m[2, 0]) / s
|
||||
x = (m[0, 1] + m[1, 0]) / s
|
||||
y = 0.25 * s
|
||||
z = (m[1, 2] + m[2, 1]) / s
|
||||
else:
|
||||
s = torch.sqrt(1.0 + m[2, 2] - m[0, 0] - m[1, 1]) * 2.0
|
||||
w = (m[1, 0] - m[0, 1]) / s
|
||||
x = (m[0, 2] + m[2, 0]) / s
|
||||
y = (m[1, 2] + m[2, 1]) / s
|
||||
z = 0.25 * s
|
||||
quats.append(torch.stack([w, x, y, z]))
|
||||
q = torch.stack(quats, dim=0)
|
||||
return q / q.norm(dim=-1, keepdim=True).clamp(min=1e-12)
|
||||
|
||||
|
||||
def _quat_wxyz_to_rotmat(q: torch.Tensor) -> torch.Tensor:
|
||||
"""Convert unit quaternions [N,4] (wxyz) to rotation matrices [N,3,3]."""
|
||||
q = q / q.norm(dim=-1, keepdim=True).clamp(min=1e-12)
|
||||
w, x, y, z = q.unbind(-1)
|
||||
R = torch.stack([
|
||||
1 - 2 * (y * y + z * z), 2 * (x * y - w * z), 2 * (x * z + w * y),
|
||||
2 * (x * y + w * z), 1 - 2 * (x * x + z * z), 2 * (y * z - w * x),
|
||||
2 * (x * z - w * y), 2 * (y * z + w * x), 1 - 2 * (x * x + y * y),
|
||||
], dim=-1).reshape(*q.shape[:-1], 3, 3)
|
||||
return R
|
||||
|
||||
|
||||
def _quat_slerp(q0: torch.Tensor, q1: torch.Tensor, alpha: torch.Tensor) -> torch.Tensor:
|
||||
"""
|
||||
Spherical linear interpolation between quaternion batches q0, q1 [N,4] (wxyz)
|
||||
with per-element interpolation factors alpha [N]. Falls back to normalized
|
||||
lerp when the quaternions are nearly parallel.
|
||||
"""
|
||||
dot = (q0 * q1).sum(dim=-1, keepdim=True)
|
||||
q1 = torch.where(dot < 0.0, -q1, q1) # shortest arc
|
||||
dot = dot.abs().clamp(max=1.0)
|
||||
a = alpha.reshape(-1, 1).to(q0.dtype)
|
||||
theta = torch.acos(dot)
|
||||
sin_theta = torch.sin(theta)
|
||||
near_parallel = sin_theta < 1e-6
|
||||
denom = sin_theta.clamp(min=1e-12)
|
||||
w0 = torch.where(near_parallel, 1.0 - a, torch.sin((1.0 - a) * theta) / denom)
|
||||
w1 = torch.where(near_parallel, a, torch.sin(a * theta) / denom)
|
||||
q = w0 * q0 + w1 * q1
|
||||
return q / q.norm(dim=-1, keepdim=True).clamp(min=1e-12)
|
||||
|
||||
|
||||
def interpolate_se3(trajectory: torch.Tensor, num_steps: int) -> torch.Tensor:
|
||||
"""trajectory [K,4,4] -> [num_steps,4,4]. Piecewise: quaternion SLERP on R, lerp on t.
|
||||
K==1 -> repeat. Must return valid rotation matrices (orthonormal)."""
|
||||
if isinstance(trajectory, np.ndarray):
|
||||
trajectory = torch.from_numpy(trajectory)
|
||||
trajectory = trajectory.float()
|
||||
if trajectory.dim() == 2:
|
||||
trajectory = trajectory.unsqueeze(0)
|
||||
if trajectory.dim() != 3 or trajectory.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"interpolate_se3 expects trajectory of shape [K,4,4], got {tuple(trajectory.shape)}")
|
||||
if num_steps < 1:
|
||||
raise ValueError(f"interpolate_se3 requires num_steps >= 1, got {num_steps}")
|
||||
K = trajectory.shape[0]
|
||||
if K == 1:
|
||||
return trajectory.expand(num_steps, 4, 4).clone()
|
||||
|
||||
R = trajectory[:, :3, :3]
|
||||
t = trajectory[:, :3, 3]
|
||||
q = _rotmat_to_quat_wxyz(R)
|
||||
# Enforce hemisphere continuity along the waypoint sequence so piecewise
|
||||
# SLERP always takes the shortest arc between consecutive poses.
|
||||
for k in range(1, K):
|
||||
if (q[k] * q[k - 1]).sum() < 0.0:
|
||||
q[k] = -q[k]
|
||||
|
||||
idxs = torch.linspace(0, K - 1, num_steps, device=trajectory.device)
|
||||
lower = idxs.floor().long().clamp(max=K - 2)
|
||||
upper = lower + 1
|
||||
alpha = (idxs - lower.float())
|
||||
|
||||
q_interp = _quat_slerp(q[lower], q[upper], alpha)
|
||||
t_interp = t[lower] * (1.0 - alpha).unsqueeze(-1) + t[upper] * alpha.unsqueeze(-1)
|
||||
|
||||
out = torch.eye(4, dtype=trajectory.dtype, device=trajectory.device).repeat(num_steps, 1, 1)
|
||||
out[:, :3, :3] = _quat_wxyz_to_rotmat(q_interp)
|
||||
out[:, :3, 3] = t_interp
|
||||
return out
|
||||
|
||||
# ==== Node Definitions ==== #
|
||||
class DepthToPointCloud:
|
||||
"""
|
||||
@@ -325,141 +438,165 @@ class ProjectPointCloud:
|
||||
|
||||
def project_pointcloud(
|
||||
self,
|
||||
pointcloud: torch.Tensor,
|
||||
pointcloud: torch.Tensor,
|
||||
output_projection: str,
|
||||
output_horizontal_fov: float,
|
||||
output_width: int,
|
||||
output_width: int,
|
||||
output_height: int,
|
||||
point_size: int = 1,
|
||||
point_size: int = 1,
|
||||
return_inverse_depth: bool = False,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
"""
|
||||
Projects an (N×6) XYZRGB point cloud into an image,
|
||||
fills occlusion holes robustly, and returns:
|
||||
• img: [1,H,W,3] RGB image
|
||||
• mask: [H,W] foreground mask
|
||||
• depth: [1,H,W,1] depth (or inverse depth)
|
||||
"""
|
||||
device = pointcloud.device
|
||||
coords = pointcloud[:, :3]
|
||||
colors = pointcloud[:, 3:].float()
|
||||
xyz, rgb_raw = pointcloud[:, :3], pointcloud[:, 3:6].float()
|
||||
|
||||
# 1) Filter points in front of the camera
|
||||
mask_front = coords[:, 2] > 0
|
||||
coords = coords[mask_front]
|
||||
colors = colors[mask_front]
|
||||
# 1) Keep only points in front of camera
|
||||
in_front = xyz[:, 2] > 0
|
||||
xyz, rgb_raw = xyz[in_front], rgb_raw[in_front]
|
||||
|
||||
# 2) Project to normalized UV + depth
|
||||
X, Y, Z = coords.unbind(1)
|
||||
X, Y, Z = xyz.unbind(1)
|
||||
if output_projection == "PINHOLE":
|
||||
u, v, depth = XYZ_to_pinhole(X, Y, Z, output_horizontal_fov)
|
||||
u, v, d = XYZ_to_pinhole(X, Y, Z, output_horizontal_fov)
|
||||
elif output_projection == "FISHEYE":
|
||||
u, v, depth = XYZ_to_fisheye(X, Y, Z, output_horizontal_fov)
|
||||
u, v, d = XYZ_to_fisheye(X, Y, Z, output_horizontal_fov)
|
||||
else:
|
||||
u, v, depth = XYZ_to_equirect(X, Y, Z, output_horizontal_fov)
|
||||
u, v, d = XYZ_to_equirect(X, Y, Z, output_horizontal_fov)
|
||||
|
||||
# 3) Rasterize to pixel indices
|
||||
px = (u * (output_width - 1) / 2) + (output_width - 1) / 2
|
||||
py = (v * (output_height - 1) / 2) + (output_height - 1) / 2
|
||||
ix = px.round().clamp(0, output_width - 1).long()
|
||||
iy = py.round().clamp(0, output_height - 1).long()
|
||||
pix = iy * output_width + ix
|
||||
M = output_width * output_height
|
||||
W, H = output_width, output_height
|
||||
ix = ((u * 0.5 + 0.5) * (W - 1)).round().clamp(0, W - 1).long()
|
||||
iy = ((v * 0.5 + 0.5) * (H - 1)).round().clamp(0, H - 1).long()
|
||||
pix = iy * W + ix
|
||||
valid = (pix >= 0) & (pix < W * H)
|
||||
pix, d, rgb_raw = pix[valid], d[valid], rgb_raw[valid]
|
||||
|
||||
# —— NEW: drop any invalid / NaN→int_min projections ——
|
||||
valid = (pix >= 0) & (pix < M)
|
||||
depth = depth[valid]
|
||||
colors = colors[valid]
|
||||
pix = pix[valid]
|
||||
order = torch.arange(depth.size(0), device=device)
|
||||
# rebuild your "order" to match
|
||||
M = W * H
|
||||
# 3a) Front‐layer (nearest) depth
|
||||
z1 = torch.full((M,), float('inf'), device=device)
|
||||
z1.scatter_reduce_(0, pix, d, reduce='amin', include_self=True)
|
||||
|
||||
# 4) Allocate or reuse buffers
|
||||
if not hasattr(self, '_z_front') or self._z_front.numel() != M:
|
||||
self._z_front = torch.empty((M,), device=device)
|
||||
self._z_back = torch.empty((M,), device=device)
|
||||
self._idx = torch.full((M,), -1, dtype=torch.long, device=device)
|
||||
self._flat = torch.zeros((M, 4), device=device)
|
||||
z_front = self._z_front
|
||||
z_back = self._z_back
|
||||
idxbuf = self._idx
|
||||
flat = self._flat
|
||||
# 3b) Second‐layer (background) depth
|
||||
farther = d > z1[pix]
|
||||
pix2, d2 = pix[farther], d[farther]
|
||||
z2 = torch.full((M,), float('inf'), device=device)
|
||||
z2.scatter_reduce_(0, pix2, d2, reduce='amin', include_self=True)
|
||||
|
||||
# 5) Front z-buffer pass (nearest)
|
||||
z_front.fill_(float('inf'))
|
||||
z_front.scatter_reduce_(0, pix, depth, reduce='amin', include_self=True)
|
||||
sel_front = depth == z_front[pix]
|
||||
order = torch.arange(depth.size(0), device=device)
|
||||
order_m = torch.where(sel_front, order, depth.size(0))
|
||||
idxbuf.fill_(depth.size(0))
|
||||
idxbuf.scatter_reduce_(0, pix, order_m, reduce='amin', include_self=True)
|
||||
win_front = order == idxbuf[pix]
|
||||
# 3c) Foreground colour (from z1)
|
||||
keep = d == z1[pix]
|
||||
rgb = torch.zeros((M, 3), device=device)
|
||||
rgb[pix[keep]] = rgb_raw[keep].clamp(0, 255)
|
||||
|
||||
flat.fill_(0)
|
||||
flat[pix[win_front]] = colors[win_front]
|
||||
img4 = flat.view(output_height, output_width, 4)
|
||||
rgb = img4[..., :3].clamp(0, 255)
|
||||
alpha = (img4[..., 3] > 0).float()
|
||||
mask_init= (img4[..., 3] > 0) # initial mask
|
||||
rgb *= alpha.unsqueeze(-1)
|
||||
depth_img = z_front.view(output_height, output_width)
|
||||
rgb_HR = rgb
|
||||
# reshape to image
|
||||
rgb = rgb.view(H, W, 3)
|
||||
z1 = z1.view(H, W)
|
||||
z2 = z2.view(H, W)
|
||||
fg_mask = z1 < float('inf') # has front hit
|
||||
occl = (z2 < float('inf')) # has any back hit
|
||||
rear_only = occl & ~fg_mask # true holes
|
||||
|
||||
# 6) Back z-buffer pass (farthest) for hole-filling
|
||||
# ── Iterative ring‐based in‐painting of rear‐only pixels ─────────────────────
|
||||
ker3 = torch.ones((1,1,3,3), device=device)
|
||||
ker3c = ker3.repeat(3,1,1,1)
|
||||
for _ in range(max(W, H)):
|
||||
# find rear_only pixels adjacent to current FG
|
||||
neigh = (
|
||||
F.max_pool2d(fg_mask.float()[None,None], 3, 1, 1).bool()[0,0]
|
||||
& ~fg_mask
|
||||
)
|
||||
to_fill = rear_only & neigh
|
||||
if not to_fill.any():
|
||||
break
|
||||
|
||||
# average depth + colour from current FG frontier
|
||||
d_t = z1.masked_fill(~fg_mask, 0)[None,None]
|
||||
c_t = rgb.permute(2,0,1)[None] # [1,3,H,W]
|
||||
m_t = fg_mask.float()[None,None]
|
||||
|
||||
sum_d = F.conv2d(d_t * m_t, ker3, padding=1)
|
||||
cnt_d = F.conv2d(m_t, ker3, padding=1).clamp(min=1)
|
||||
sum_c = F.conv2d(c_t * m_t, ker3c, padding=1, groups=3)
|
||||
cnt_c = cnt_d.repeat(1,3,1,1)
|
||||
|
||||
avg_d = (sum_d / cnt_d).squeeze()
|
||||
avg_c = (sum_c / cnt_c).squeeze().permute(1,2,0)
|
||||
|
||||
z1[to_fill] = avg_d[to_fill]
|
||||
rgb[to_fill] = avg_c[to_fill]
|
||||
fg_mask[to_fill] = True
|
||||
rear_only[to_fill] = False
|
||||
|
||||
# ── Depth‐aware generic hole closure ─────────────────────────────────────────
|
||||
# close_rad: radius of hole to close; depth_eps: depth jump tolerance
|
||||
close_rad = max(1, point_size // 2)
|
||||
depth_eps = 0.015
|
||||
pad = close_rad
|
||||
k = 2 * close_rad + 1
|
||||
ker = torch.ones((1,1,k,k), device=device)
|
||||
kerc = ker.repeat(3,1,1,1)
|
||||
front_t = fg_mask.float()[None,None]
|
||||
|
||||
# binary closing: dilate then erode
|
||||
D = F.max_pool2d(front_t, k, 1, pad)
|
||||
E = 1 - F.max_pool2d(1 - D, k, 1, pad)
|
||||
small_hole = E[0,0].bool() & ~fg_mask
|
||||
if small_hole.any():
|
||||
# compute local mean depth of FG
|
||||
z_t = z1.masked_fill(~fg_mask, 0)[None,None]
|
||||
cnt = F.conv2d(front_t, ker, padding=pad).clamp(min=1)
|
||||
z_avg = (F.conv2d(z_t, ker, padding=pad) / cnt)[0,0]
|
||||
|
||||
# depth‐range test
|
||||
z_near = F.max_pool2d(z1[None,None], 3,1,1)[0,0]
|
||||
z_far = -F.max_pool2d(-z1[None,None],3,1,1)[0,0]
|
||||
flat = (z_far - z_near) / z_avg.clamp(min=1e-6) < depth_eps
|
||||
|
||||
final = small_hole & flat
|
||||
if final.any():
|
||||
sum_d = F.conv2d(z_t, ker, padding=pad)
|
||||
sum_c = F.conv2d(rgb.permute(2,0,1)[None] * front_t, kerc,
|
||||
padding=pad, groups=3)
|
||||
avg_d = (sum_d / cnt)[0,0]
|
||||
avg_c = (sum_c / cnt.repeat(1,3,1,1))[0].permute(1,2,0)
|
||||
|
||||
z1[final] = avg_d[final]
|
||||
rgb[final] = avg_c[final]
|
||||
fg_mask[final] = True
|
||||
|
||||
# ── Optional morphological blur for larger point_size ───────────────────────
|
||||
if point_size > 1:
|
||||
z_back.fill_(-float('inf'))
|
||||
z_back.scatter_reduce_(0, pix, depth, reduce='amax', include_self=True)
|
||||
sel_back = depth == z_back[pix]
|
||||
order_m = torch.where(sel_back, order, -1)
|
||||
idxbuf.fill_(-1)
|
||||
idxbuf.scatter_reduce_(0, pix, order_m, reduce='amax', include_self=True)
|
||||
win_back = idxbuf[pix] >= 0
|
||||
flat.fill_(0)
|
||||
flat[pix[win_back]] = colors[win_back]
|
||||
back4 = flat.view(output_height, output_width, 4)
|
||||
rgb_back = back4[..., :3].clamp(0,255)
|
||||
alpha_back = (back4[..., 3] > 0).float()
|
||||
r = point_size // 2
|
||||
k = 2 * r + 1
|
||||
pad = r
|
||||
ker = torch.ones((1,1,k,k), device=device)
|
||||
kerc = ker.repeat(3,1,1,1)
|
||||
d_t = z1[None,None]
|
||||
c_t = rgb.permute(2,0,1)[None]
|
||||
m_t = fg_mask.float()[None,None]
|
||||
|
||||
# fill holes where front missed
|
||||
hole = (alpha == 0) & (alpha_back > 0)
|
||||
rgb[hole] = rgb_back[hole]
|
||||
alpha[hole] = 1.0
|
||||
depth_img[hole] = z_back.view(output_height, output_width)[hole]
|
||||
z1 = (F.conv2d(d_t * m_t, ker, padding=pad) /
|
||||
F.conv2d(m_t, ker, padding=pad).clamp(min=1)).squeeze()
|
||||
rgb = (F.conv2d(c_t * m_t, kerc, padding=pad, groups=3) /
|
||||
F.conv2d(m_t, ker, padding=pad).repeat(1,3,1,1).clamp(min=1)
|
||||
).squeeze().permute(1,2,0)
|
||||
|
||||
# 7) Median-filter _only_ in hole regions
|
||||
if hole.any():
|
||||
# prepare for kornia median_blur: [B,C,H,W]
|
||||
rgb_t = rgb.permute(2,0,1).unsqueeze(0) # [1,3,H,W]
|
||||
# apply median filter
|
||||
rgb_med = median_blur(rgb_t, (point_size, point_size))
|
||||
# back to HWC
|
||||
rgb_med = rgb_med.squeeze(0).permute(1,2,0)
|
||||
# merge only at hole locations
|
||||
rgb[hole] = rgb_med[hole]
|
||||
# alpha already set to 1.0 for holes
|
||||
# 8 apply median blur to mask if point_size > 1 and to initial image
|
||||
mask_t = alpha.unsqueeze(0).unsqueeze(0) # [1,1,H,W]
|
||||
pad = point_size // 2
|
||||
ksize = (point_size, point_size)
|
||||
|
||||
# b) grow (dilate) mask by max‑pool
|
||||
mask_grow = F.max_pool2d(mask_t, kernel_size=ksize, stride=1, padding=pad)
|
||||
|
||||
# c) shrink (erode) by inverting, max‑pool, then inverting back
|
||||
mask_shrink = 1.0 - F.max_pool2d(1.0 - mask_grow, kernel_size=ksize, stride=1, padding=pad)
|
||||
|
||||
# d) back to [H,W] and use as our new alpha
|
||||
alpha = mask_shrink.squeeze(0).squeeze(0)
|
||||
print(1)
|
||||
# e) median‑filter the *whole* RGB image
|
||||
# prep for kornia: [B,C,H,W]
|
||||
rgb_t_full = rgb.permute(2,0,1).unsqueeze(0) # [1,3,H,W]
|
||||
rgb_med_full = median_blur(rgb_t_full, ksize) # [1,3,H,W]
|
||||
rgb_med_full = rgb_med_full.squeeze(0).permute(1,2,0) # [H,W,3]
|
||||
|
||||
# g) refill *only* the original holes with the median result
|
||||
rgb[~mask_init] = rgb_med_full[~mask_init]
|
||||
# 9) Pack and return with original script shapes
|
||||
img = rgb.unsqueeze(0) # [1,H,W,3]
|
||||
mask_out = alpha # [H,W]
|
||||
depth4 = depth_img.unsqueeze(0).unsqueeze(-1) # [1,H,W,1]
|
||||
# 10) Pack outputs
|
||||
img = rgb.unsqueeze(0) # [1,H,W,3]
|
||||
mask = fg_mask.float() # [H,W]
|
||||
depth = z1.unsqueeze(0).unsqueeze(-1) # [1,H,W,1]
|
||||
if return_inverse_depth:
|
||||
depth4 = 1.0 / depth4.clamp(min=1e-6)
|
||||
depth4 = depth4 * mask_out.unsqueeze(0).unsqueeze(-1)
|
||||
return img, mask_out, depth4
|
||||
depth = 1.0 / depth.clamp(min=1e-6)
|
||||
depth *= mask.unsqueeze(0).unsqueeze(-1)
|
||||
return img, mask, depth
|
||||
|
||||
|
||||
|
||||
|
||||
class PointCloudUnion:
|
||||
"""
|
||||
@@ -770,8 +907,9 @@ class CameraMotionNode:
|
||||
|
||||
class CameraInterpolationNode:
|
||||
"""
|
||||
Wrap two 4×4 poses into a trajectory tensor.
|
||||
Outputs only `trajectory` (shape 2×4×4).
|
||||
Interpolate between two 4×4 poses into a trajectory tensor using proper
|
||||
SE(3) interpolation (quaternion SLERP on rotation, lerp on translation).
|
||||
Outputs `trajectory` (shape num_steps×4×4, default 2×4×4).
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
@@ -780,7 +918,10 @@ class CameraInterpolationNode:
|
||||
"required": {
|
||||
"initial_matrix": ("MAT_4X4",),
|
||||
"final_matrix": ("MAT_4X4",),
|
||||
}
|
||||
},
|
||||
"optional": {
|
||||
"num_steps": ("INT", {"default": 2, "min": 2, "max": 4096, "tooltip": "Number of poses in the output trajectory, SE(3)-interpolated between the two matrices."}),
|
||||
},
|
||||
}
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("trajectory",)
|
||||
@@ -791,14 +932,15 @@ class CameraInterpolationNode:
|
||||
self,
|
||||
initial_matrix: torch.Tensor,
|
||||
final_matrix: torch.Tensor,
|
||||
num_steps: int = 2,
|
||||
) -> Tuple[torch.Tensor]:
|
||||
# stack into a (2,4,4) trajectory
|
||||
# convert to tensor if needed
|
||||
if isinstance(initial_matrix, np.ndarray):
|
||||
initial_matrix = torch.from_numpy(initial_matrix).float()
|
||||
if isinstance(final_matrix, np.ndarray):
|
||||
final_matrix = torch.from_numpy(final_matrix).float()
|
||||
traj = torch.stack([initial_matrix, final_matrix], dim=0)
|
||||
keyframes = torch.stack([initial_matrix.float(), final_matrix.float()], dim=0)
|
||||
traj = interpolate_se3(keyframes, num_steps)
|
||||
return (traj,)
|
||||
|
||||
|
||||
@@ -813,7 +955,7 @@ class CameraTrajectoryNode:
|
||||
"pointcloud": ("TENSOR",),
|
||||
},
|
||||
"optional": {
|
||||
"initial_matrix": ("MAT_4X4"),
|
||||
"initial_matrix": ("MAT_4X4",),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1226,6 +1368,95 @@ class LoadTrajectory:
|
||||
return f"Invalid trajectory file: {trajectory_file}"
|
||||
return True
|
||||
|
||||
class DepthEdgeFilter:
|
||||
"""
|
||||
Detect "flying pixel" depth discontinuities and output a validity mask.
|
||||
A pixel is flagged as an edge where |depth gradient| / depth exceeds
|
||||
`relative_threshold`; edges are optionally dilated. Returns a MASK with
|
||||
1.0 where the depth is valid (NOT a flying-pixel edge) and 0.0 on edges.
|
||||
"""
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
# Depth: [H,W] or [T,H,W], trailing channel dim of 1 accepted
|
||||
"depth": ("TENSOR", {"shape_hint": [None, None, None]}),
|
||||
"relative_threshold": ("FLOAT", {"default": 0.05, "min": 0.0, "max": 10.0, "step": 0.005, "tooltip": "Mark a pixel as edge where |depth gradient| / depth exceeds this value."}),
|
||||
"dilate": ("INT", {"default": 1, "min": 0, "max": 64, "tooltip": "Grow detected edges by this many pixels (max-pool dilation)."}),
|
||||
},
|
||||
"optional": {
|
||||
"mask": ("MASK", {"tooltip": "Optional validity mask ANDed with the edge-filter result."}),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("MASK",)
|
||||
RETURN_NAMES = ("valid_mask",)
|
||||
FUNCTION = "filter_edges"
|
||||
CATEGORY = "Camera/PointCloud"
|
||||
|
||||
def filter_edges(
|
||||
self,
|
||||
depth: torch.Tensor,
|
||||
relative_threshold: float,
|
||||
dilate: int,
|
||||
mask: torch.Tensor = None,
|
||||
) -> Tuple[torch.Tensor]:
|
||||
d = depth
|
||||
if isinstance(d, np.ndarray):
|
||||
d = torch.from_numpy(d)
|
||||
d = d.float()
|
||||
# Accept [H,W], [H,W,1], [T,H,W], [T,H,W,1]
|
||||
if d.dim() == 4 and d.shape[-1] == 1:
|
||||
d = d[..., 0]
|
||||
elif d.dim() == 3 and d.shape[-1] == 1:
|
||||
d = d[..., 0]
|
||||
squeeze_batch = False
|
||||
if d.dim() == 2:
|
||||
d = d.unsqueeze(0)
|
||||
squeeze_batch = True
|
||||
if d.dim() != 3:
|
||||
raise ValueError(f"DepthEdgeFilter expects depth of shape [H,W] or [T,H,W] (trailing 1 ok), got {tuple(depth.shape)}")
|
||||
|
||||
eps = 1e-8
|
||||
# Forward differences along x and y; propagate each difference to both
|
||||
# neighbouring pixels so both sides of a discontinuity are flagged.
|
||||
dx = (d[:, :, 1:] - d[:, :, :-1]).abs()
|
||||
dy = (d[:, 1:, :] - d[:, :-1, :]).abs()
|
||||
gx = torch.zeros_like(d)
|
||||
gx[:, :, :-1] = dx
|
||||
gx[:, :, 1:] = torch.maximum(gx[:, :, 1:], dx)
|
||||
gy = torch.zeros_like(d)
|
||||
gy[:, :-1, :] = dy
|
||||
gy[:, 1:, :] = torch.maximum(gy[:, 1:, :], dy)
|
||||
grad = torch.maximum(gx, gy)
|
||||
edge = (grad / d.abs().clamp(min=eps)) > relative_threshold
|
||||
|
||||
if dilate > 0:
|
||||
k = 2 * int(dilate) + 1
|
||||
edge = F.max_pool2d(edge.float().unsqueeze(1), kernel_size=k, stride=1, padding=int(dilate)).squeeze(1) > 0.5
|
||||
|
||||
valid = (~edge).float()
|
||||
|
||||
if mask is not None:
|
||||
m = mask
|
||||
if isinstance(m, np.ndarray):
|
||||
m = torch.from_numpy(m)
|
||||
m = m.float().to(valid.device)
|
||||
if m.dim() == 4 and m.shape[-1] == 1:
|
||||
m = m[..., 0]
|
||||
if m.dim() == 2:
|
||||
m = m.unsqueeze(0)
|
||||
if m.shape[0] == 1 and valid.shape[0] > 1:
|
||||
m = m.expand(valid.shape[0], -1, -1)
|
||||
if m.shape[-2:] != valid.shape[-2:]:
|
||||
m = F.interpolate(m.unsqueeze(1), size=valid.shape[-2:], mode="nearest").squeeze(1)
|
||||
valid = valid * (m > 0.5).float()
|
||||
|
||||
if squeeze_batch:
|
||||
valid = valid[0]
|
||||
return (valid,)
|
||||
|
||||
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"DepthToPointCloud": DepthToPointCloud,
|
||||
"TransformPointCloud": TransformPointCloud,
|
||||
@@ -1240,4 +1471,5 @@ NODE_CLASS_MAPPINGS = {
|
||||
"PointCloudCleaner": PointCloudCleaner,
|
||||
"SaveTrajectory": SaveTrajectory,
|
||||
"LoadTrajectory": LoadTrajectory,
|
||||
"DepthEdgeFilter": DepthEdgeFilter,
|
||||
}
|
||||
@@ -0,0 +1,462 @@
|
||||
"""Camera pose estimation nodes.
|
||||
|
||||
Provides:
|
||||
- VideoPoseEstimator: VGGT-based per-frame camera pose + depth + intrinsics
|
||||
estimation from a video clip.
|
||||
- TrajectoryInvert / TrajectoryCompose: small utility nodes for wiring
|
||||
trajectory tensors ([K, 4, 4] world-to-camera matrices) in graphs.
|
||||
|
||||
Coordinate convention (matches the rest of this repo): camera frame is
|
||||
+X right, +Y down, +Z forward; trajectory matrices are 4x4 world-to-camera
|
||||
(`cam_pts = world_pts @ R.T + t`). VGGT outputs OpenCV-convention
|
||||
camera-from-world extrinsics, which match this convention directly.
|
||||
"""
|
||||
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
from tqdm import tqdm
|
||||
|
||||
try:
|
||||
import folder_paths
|
||||
except ImportError: # Allow notebook usage outside ComfyUI
|
||||
class _FolderPathsStub:
|
||||
def __getattr__(self, name):
|
||||
raise ModuleNotFoundError(
|
||||
"folder_paths is unavailable; this feature requires the ComfyUI runtime."
|
||||
)
|
||||
|
||||
folder_paths = _FolderPathsStub()
|
||||
|
||||
_here = os.path.dirname(os.path.abspath(__file__))
|
||||
# climb up 2 levels: camera-comfyUI -> custom_nodes -> ComfyUI
|
||||
COMFYUI_ROOT = os.path.abspath(os.path.join(_here, os.pardir, os.pardir))
|
||||
|
||||
DEVICE_CHOICES = ["auto", "cpu", "cuda"]
|
||||
|
||||
# Module-level model cache: {device_str: model}
|
||||
_VGGT_MODEL_CACHE: Dict[str, Any] = {}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# SE(3) interpolation (contract C1). Prefer the shared implementation from
|
||||
# pointcloud_nodes; fall back to a local copy so this file works standalone.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def _matrix_to_quaternion(R: torch.Tensor) -> torch.Tensor:
|
||||
"""Convert a single 3x3 rotation matrix to a wxyz quaternion."""
|
||||
R = R.to(torch.float64)
|
||||
m00, m01, m02 = R[0, 0], R[0, 1], R[0, 2]
|
||||
m10, m11, m12 = R[1, 0], R[1, 1], R[1, 2]
|
||||
m20, m21, m22 = R[2, 0], R[2, 1], R[2, 2]
|
||||
trace = m00 + m11 + m22
|
||||
if trace > 0.0:
|
||||
s = torch.sqrt(trace + 1.0) * 2.0
|
||||
w = 0.25 * s
|
||||
x = (m21 - m12) / s
|
||||
y = (m02 - m20) / s
|
||||
z = (m10 - m01) / s
|
||||
elif (m00 > m11) and (m00 > m22):
|
||||
s = torch.sqrt(1.0 + m00 - m11 - m22) * 2.0
|
||||
w = (m21 - m12) / s
|
||||
x = 0.25 * s
|
||||
y = (m01 + m10) / s
|
||||
z = (m02 + m20) / s
|
||||
elif m11 > m22:
|
||||
s = torch.sqrt(1.0 + m11 - m00 - m22) * 2.0
|
||||
w = (m02 - m20) / s
|
||||
x = (m01 + m10) / s
|
||||
y = 0.25 * s
|
||||
z = (m12 + m21) / s
|
||||
else:
|
||||
s = torch.sqrt(1.0 + m22 - m00 - m11) * 2.0
|
||||
w = (m10 - m01) / s
|
||||
x = (m02 + m20) / s
|
||||
y = (m12 + m21) / s
|
||||
z = 0.25 * s
|
||||
q = torch.stack([w, x, y, z])
|
||||
return (q / q.norm().clamp(min=1e-12)).to(torch.float32)
|
||||
|
||||
|
||||
def _quaternion_to_matrix(q: torch.Tensor) -> torch.Tensor:
|
||||
"""Convert a wxyz quaternion to a 3x3 rotation matrix."""
|
||||
q = q / q.norm().clamp(min=1e-12)
|
||||
w, x, y, z = q[0], q[1], q[2], q[3]
|
||||
return torch.stack([
|
||||
torch.stack([1 - 2 * (y * y + z * z), 2 * (x * y - w * z), 2 * (x * z + w * y)]),
|
||||
torch.stack([2 * (x * y + w * z), 1 - 2 * (x * x + z * z), 2 * (y * z - w * x)]),
|
||||
torch.stack([2 * (x * z - w * y), 2 * (y * z + w * x), 1 - 2 * (x * x + y * y)]),
|
||||
])
|
||||
|
||||
|
||||
def _quat_slerp(q0: torch.Tensor, q1: torch.Tensor, alpha: float) -> torch.Tensor:
|
||||
"""Spherical linear interpolation between two wxyz quaternions."""
|
||||
q0 = q0 / q0.norm().clamp(min=1e-12)
|
||||
q1 = q1 / q1.norm().clamp(min=1e-12)
|
||||
dot = torch.dot(q0, q1)
|
||||
if dot < 0.0: # take the short path on the quaternion hypersphere
|
||||
q1 = -q1
|
||||
dot = -dot
|
||||
dot = dot.clamp(-1.0, 1.0)
|
||||
if dot > 0.9995: # nearly parallel: lerp + renormalize is numerically safer
|
||||
q = (1.0 - alpha) * q0 + alpha * q1
|
||||
return q / q.norm().clamp(min=1e-12)
|
||||
theta = torch.acos(dot)
|
||||
sin_theta = torch.sin(theta)
|
||||
w0 = torch.sin((1.0 - alpha) * theta) / sin_theta
|
||||
w1 = torch.sin(alpha * theta) / sin_theta
|
||||
q = w0 * q0 + w1 * q1
|
||||
return q / q.norm().clamp(min=1e-12)
|
||||
|
||||
|
||||
def _interpolate_se3_fallback(trajectory: torch.Tensor, num_steps: int) -> torch.Tensor:
|
||||
"""trajectory [K,4,4] -> [num_steps,4,4]. Piecewise: quaternion SLERP on R, lerp on t.
|
||||
K==1 -> repeat. Returns valid (orthonormal) rotation matrices. Matches contract C1."""
|
||||
traj = torch.as_tensor(trajectory, dtype=torch.float32)
|
||||
if traj.dim() == 2:
|
||||
traj = traj.unsqueeze(0)
|
||||
if traj.dim() != 3 or traj.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"Expected trajectory of shape [K,4,4], got {tuple(traj.shape)}")
|
||||
K = traj.shape[0]
|
||||
if K == 1:
|
||||
return traj.expand(num_steps, 4, 4).clone()
|
||||
quats = torch.stack([_matrix_to_quaternion(traj[i, :3, :3]) for i in range(K)])
|
||||
trans = traj[:, :3, 3]
|
||||
positions = torch.linspace(0.0, float(K - 1), num_steps)
|
||||
out = []
|
||||
for pos in positions:
|
||||
lower = int(torch.floor(pos).clamp(max=K - 2))
|
||||
upper = lower + 1
|
||||
alpha = float(pos) - lower
|
||||
q = _quat_slerp(quats[lower], quats[upper], alpha)
|
||||
t = (1.0 - alpha) * trans[lower] + alpha * trans[upper]
|
||||
M = torch.eye(4, dtype=torch.float32)
|
||||
M[:3, :3] = _quaternion_to_matrix(q)
|
||||
M[:3, 3] = t
|
||||
out.append(M)
|
||||
return torch.stack(out, dim=0)
|
||||
|
||||
|
||||
try:
|
||||
from .pointcloud_nodes import interpolate_se3
|
||||
except Exception:
|
||||
try:
|
||||
from pointcloud_nodes import interpolate_se3
|
||||
except Exception:
|
||||
interpolate_se3 = _interpolate_se3_fallback
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# VGGT lazy import helpers
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
def _import_vggt() -> Tuple[Any, Any]:
|
||||
"""Lazily import VGGT. Tries the pip package first, then a sibling clone
|
||||
at COMFYUI_ROOT/vggt (mirroring how video_nodes.py handles Video-Depth-Anything)."""
|
||||
try:
|
||||
from vggt.models.vggt import VGGT
|
||||
from vggt.utils.pose_enc import pose_encoding_to_extri_intri
|
||||
return VGGT, pose_encoding_to_extri_intri
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
vggt_clone_path = os.path.join(COMFYUI_ROOT, "vggt")
|
||||
if os.path.isdir(vggt_clone_path) and vggt_clone_path not in sys.path:
|
||||
sys.path.insert(0, vggt_clone_path)
|
||||
try:
|
||||
from vggt.models.vggt import VGGT
|
||||
from vggt.utils.pose_enc import pose_encoding_to_extri_intri
|
||||
return VGGT, pose_encoding_to_extri_intri
|
||||
except ImportError as exc:
|
||||
raise ModuleNotFoundError(
|
||||
"VGGT is not installed. Run this pack's install.py (ComfyUI-Manager does this "
|
||||
"automatically), or install it manually with "
|
||||
"`pip install git+https://github.com/facebookresearch/vggt.git` (it is not on PyPI), "
|
||||
f"or clone https://github.com/facebookresearch/vggt into {vggt_clone_path!r}. "
|
||||
"It also requires `huggingface_hub` to download the facebook/VGGT-1B weights."
|
||||
) from exc
|
||||
|
||||
|
||||
def _get_vggt_model(device: torch.device) -> Any:
|
||||
"""Load (and cache) the VGGT-1B model on the requested device."""
|
||||
key = str(device)
|
||||
if key not in _VGGT_MODEL_CACHE:
|
||||
VGGT, _ = _import_vggt()
|
||||
print(f"[pose_nodes] Loading facebook/VGGT-1B onto {key} (first call downloads ~5GB weights)...")
|
||||
model = VGGT.from_pretrained("facebook/VGGT-1B")
|
||||
model = model.to(device).eval()
|
||||
_VGGT_MODEL_CACHE[key] = model
|
||||
return _VGGT_MODEL_CACHE[key]
|
||||
|
||||
|
||||
def _vggt_preprocess(frames: torch.Tensor, resolution: int, device: torch.device) -> torch.Tensor:
|
||||
"""[T,H,W,3] float 0..1 -> [1,T,3,Hp,Wp] with max dim == resolution (both dims
|
||||
divisible by 14, the VGGT patch size), aspect ratio preserved."""
|
||||
T, H, W, _ = frames.shape
|
||||
imgs = frames.permute(0, 3, 1, 2).to(device=device, dtype=torch.float32)
|
||||
if imgs.max() > 1.5: # defensively handle 0..255 inputs
|
||||
imgs = imgs / 255.0
|
||||
scale = float(resolution) / float(max(H, W))
|
||||
new_h = max(14, int(round(H * scale / 14.0)) * 14)
|
||||
new_w = max(14, int(round(W * scale / 14.0)) * 14)
|
||||
if (new_h, new_w) != (H, W):
|
||||
imgs = F.interpolate(imgs, size=(new_h, new_w), mode="bilinear", align_corners=False)
|
||||
return imgs.clamp(0.0, 1.0).unsqueeze(0) # [1,T,3,Hp,Wp]
|
||||
|
||||
|
||||
class VideoPoseEstimator:
|
||||
"""
|
||||
Estimates per-frame camera poses (world-to-camera [T,4,4]), metric-ish depth
|
||||
maps, depth confidence and the horizontal FOV from a video clip using
|
||||
facebook/VGGT-1B.
|
||||
|
||||
VGGT extrinsics use the OpenCV camera convention (+X right, +Y down,
|
||||
+Z forward, camera-from-world), which matches this repo's trajectory
|
||||
convention, so the matrices are returned as-is (padded to 4x4).
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
# Video frames: Tensor [T, H, W, 3] float 0..1
|
||||
"frames": ("IMAGE", {"shape_hint": [None, None, None, 3]}),
|
||||
"max_frames": ("INT", {
|
||||
"default": 64, "min": 1, "max": 1024,
|
||||
"tooltip": "If the clip has more frames than this, it is stride-subsampled "
|
||||
"for VGGT and the poses are SE(3)-interpolated back to full length "
|
||||
"(depth/confidence use nearest-frame fill).",
|
||||
}),
|
||||
"resolution": ("INT", {
|
||||
"default": 518, "min": 98, "max": 1036,
|
||||
"tooltip": "Max image dimension fed to VGGT (rounded to a multiple of 14).",
|
||||
}),
|
||||
"device": (DEVICE_CHOICES, {"default": "auto"}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("TENSOR", "TENSOR", "FLOAT", "TENSOR")
|
||||
RETURN_NAMES = ("trajectory", "depths", "horizontal_fov", "confidence")
|
||||
FUNCTION = "estimate_poses"
|
||||
CATEGORY = "Camera/Pose"
|
||||
DESCRIPTION = (
|
||||
"VGGT camera pose + depth estimation. Outputs world-to-camera trajectory [T,4,4], "
|
||||
"depth maps [T,H,W] at the input resolution, mean horizontal FOV (degrees) and "
|
||||
"per-pixel depth confidence [T,H,W]."
|
||||
)
|
||||
|
||||
def estimate_poses(
|
||||
self,
|
||||
frames: torch.Tensor,
|
||||
max_frames: int = 64,
|
||||
resolution: int = 518,
|
||||
device: str = "auto",
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, float, torch.Tensor]:
|
||||
if frames.dim() != 4 or frames.shape[-1] != 3:
|
||||
raise ValueError(f"Expected frames of shape [T,H,W,3], got {tuple(frames.shape)}")
|
||||
T_full, H, W, _ = frames.shape
|
||||
|
||||
if device == "auto":
|
||||
dev = torch.device("cuda" if torch.cuda.is_available() else "cpu")
|
||||
elif device == "cuda":
|
||||
if not torch.cuda.is_available():
|
||||
raise ValueError("CUDA requested but not available.")
|
||||
dev = torch.device("cuda")
|
||||
else:
|
||||
dev = torch.device("cpu")
|
||||
|
||||
# Stride-subsample overly long clips, keeping the frame mapping so that
|
||||
# poses can be interpolated back afterwards.
|
||||
if T_full > max_frames:
|
||||
sub_indices = torch.linspace(0, T_full - 1, max_frames).round().long().unique()
|
||||
print(
|
||||
f"[VideoPoseEstimator] WARNING: clip has {T_full} frames > max_frames={max_frames}; "
|
||||
f"running VGGT on {sub_indices.numel()} stride-subsampled frames. Poses are "
|
||||
"SE(3)-interpolated back to full length; depth/confidence use nearest-frame fill. "
|
||||
"Increase max_frames for exact per-frame estimates."
|
||||
)
|
||||
proc_frames = frames[sub_indices]
|
||||
else:
|
||||
sub_indices = None
|
||||
proc_frames = frames
|
||||
|
||||
images = _vggt_preprocess(proc_frames, resolution, dev) # [1,S,3,Hp,Wp]
|
||||
S, Hp, Wp = images.shape[1], images.shape[-2], images.shape[-1]
|
||||
|
||||
_, pose_encoding_to_extri_intri = _import_vggt()
|
||||
model = _get_vggt_model(dev)
|
||||
|
||||
try:
|
||||
with torch.no_grad():
|
||||
if dev.type == "cuda":
|
||||
capability = torch.cuda.get_device_capability(dev)
|
||||
amp_dtype = torch.bfloat16 if capability[0] >= 8 else torch.float16
|
||||
with torch.autocast(device_type="cuda", dtype=amp_dtype):
|
||||
aggregated_tokens_list, ps_idx = model.aggregator(images)
|
||||
else:
|
||||
aggregated_tokens_list, ps_idx = model.aggregator(images)
|
||||
# Camera + depth heads run in full precision (per the official VGGT example).
|
||||
pose_enc = model.camera_head(aggregated_tokens_list)[-1]
|
||||
extrinsic, intrinsic = pose_encoding_to_extri_intri(pose_enc, images.shape[-2:])
|
||||
depth_map, depth_conf = model.depth_head(aggregated_tokens_list, images, ps_idx)
|
||||
except torch.cuda.OutOfMemoryError as exc:
|
||||
raise RuntimeError(
|
||||
f"VGGT ran out of GPU memory on {S} frames at {Wp}x{Hp}. "
|
||||
"Lower max_frames and/or resolution, or set device='cpu' (slow)."
|
||||
) from exc
|
||||
|
||||
# ---- Trajectory: pad OpenCV world-to-camera [S,3,4] to [S,4,4] ---- #
|
||||
extrinsic = extrinsic.squeeze(0).to(torch.float32).cpu() # [S,3,4]
|
||||
trajectory = torch.eye(4, dtype=torch.float32).unsqueeze(0).repeat(extrinsic.shape[0], 1, 1)
|
||||
trajectory[:, :3, :4] = extrinsic
|
||||
|
||||
# ---- Horizontal FOV from intrinsics (resolution-invariant fx/W ratio) ---- #
|
||||
intrinsic = intrinsic.squeeze(0).to(torch.float32).cpu() # [S,3,3]
|
||||
fx = intrinsic[:, 0, 0].clamp(min=1e-6)
|
||||
hfov_per_frame = 2.0 * torch.atan(0.5 * float(Wp) / fx) # radians, at processing width
|
||||
# Aspect ratio is preserved during preprocessing, so fx/W is the same at
|
||||
# the original width and the FOV needs no conversion.
|
||||
horizontal_fov = float(torch.rad2deg(hfov_per_frame).mean())
|
||||
|
||||
# ---- Depth + confidence, resized back to the input resolution ---- #
|
||||
depth = depth_map.squeeze(0).to(torch.float32).cpu() # [S,Hp,Wp,1] (or [S,Hp,Wp])
|
||||
if depth.dim() == 4 and depth.shape[-1] == 1:
|
||||
depth = depth.squeeze(-1)
|
||||
conf = depth_conf.squeeze(0).to(torch.float32).cpu() # [S,Hp,Wp]
|
||||
if conf.dim() == 4 and conf.shape[-1] == 1:
|
||||
conf = conf.squeeze(-1)
|
||||
|
||||
# ---- Convert VGGT z-depth to RADIAL ray depth ---- #
|
||||
# VGGT's depth head predicts z-depth (its unprojection is
|
||||
# x = (u - cx) * d / fx, z = d), while every consumer in this repo
|
||||
# (pointcloud *_depth_to_XYZ helpers, MotionMaskFromDepth,
|
||||
# TracksToTrajectories, the GS4D helpers) multiplies unit ray directions
|
||||
# by depth, i.e. expects RADIAL distance. Multiply by the per-pixel ray
|
||||
# norm sqrt(1 + ((u-cx)/fx)^2 + ((v-cy)/fy)^2) using the per-frame
|
||||
# intrinsics at the VGGT processing resolution.
|
||||
fx_pf = intrinsic[:, 0, 0].clamp(min=1e-6).view(-1, 1, 1) # [S,1,1]
|
||||
fy_pf = intrinsic[:, 1, 1].clamp(min=1e-6).view(-1, 1, 1)
|
||||
cx_pf = intrinsic[:, 0, 2].view(-1, 1, 1)
|
||||
cy_pf = intrinsic[:, 1, 2].view(-1, 1, 1)
|
||||
uu = torch.arange(Wp, dtype=torch.float32).view(1, 1, -1)
|
||||
vv = torch.arange(Hp, dtype=torch.float32).view(1, -1, 1)
|
||||
xn = (uu - cx_pf) / fx_pf
|
||||
yn = (vv - cy_pf) / fy_pf
|
||||
depth = depth * torch.sqrt(1.0 + xn * xn + yn * yn)
|
||||
|
||||
if (Hp, Wp) != (H, W):
|
||||
depth = F.interpolate(depth.unsqueeze(1), size=(H, W), mode="bilinear", align_corners=False).squeeze(1)
|
||||
conf = F.interpolate(conf.unsqueeze(1), size=(H, W), mode="bilinear", align_corners=False).squeeze(1)
|
||||
|
||||
# ---- If subsampled, expand back to the full frame count ---- #
|
||||
if sub_indices is not None:
|
||||
# Subsample indices are (near-)uniform over [0, T_full-1], so uniform
|
||||
# SE(3) resampling reconstructs per-frame poses well.
|
||||
trajectory = interpolate_se3(trajectory, T_full)
|
||||
all_t = torch.arange(T_full).unsqueeze(1) # [T_full,1]
|
||||
nearest = (sub_indices.unsqueeze(0) - all_t).abs().argmin(dim=1) # [T_full]
|
||||
depth = depth[nearest]
|
||||
conf = conf[nearest]
|
||||
|
||||
return (trajectory, depth, horizontal_fov, conf)
|
||||
|
||||
|
||||
class TrajectoryInvert:
|
||||
"""
|
||||
Inverts each 4x4 matrix in a trajectory tensor, converting between
|
||||
world-to-camera and camera-to-world conventions.
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
# Trajectory: Tensor [K, 4, 4] (a single [4, 4] matrix also works)
|
||||
"trajectory": ("TENSOR", {"shape_hint": [None, 4, 4]}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("trajectory",)
|
||||
FUNCTION = "invert"
|
||||
CATEGORY = "Camera/Pose"
|
||||
DESCRIPTION = "Inverts each 4x4 pose (world-to-camera <-> camera-to-world)."
|
||||
|
||||
def invert(self, trajectory: torch.Tensor) -> Tuple[torch.Tensor]:
|
||||
traj = torch.as_tensor(trajectory, dtype=torch.float32)
|
||||
squeeze = traj.dim() == 2
|
||||
if squeeze:
|
||||
traj = traj.unsqueeze(0)
|
||||
if traj.dim() != 3 or traj.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"Expected trajectory of shape [K,4,4], got {tuple(trajectory.shape)}")
|
||||
# Rigid-body inverse: R -> R.T, t -> -R.T @ t (numerically stabler than
|
||||
# a generic matrix inverse for SE(3) poses).
|
||||
R = traj[:, :3, :3]
|
||||
t = traj[:, :3, 3:4]
|
||||
Rt = R.transpose(1, 2)
|
||||
inv = torch.eye(4, dtype=traj.dtype).unsqueeze(0).repeat(traj.shape[0], 1, 1)
|
||||
inv[:, :3, :3] = Rt
|
||||
inv[:, :3, 3:4] = -Rt @ t
|
||||
if squeeze:
|
||||
inv = inv.squeeze(0)
|
||||
return (inv,)
|
||||
|
||||
|
||||
class TrajectoryCompose:
|
||||
"""
|
||||
Composes two trajectories per frame: out_k = A_k @ B_k. Either input may be
|
||||
a single [4,4] matrix, which is broadcast against the other. Useful for
|
||||
retargeting novel camera paths relative to a source pose (e.g. compose a
|
||||
relative path with the inverse of source pose 0).
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
# Left operand: Tensor [K, 4, 4] or [4, 4]
|
||||
"trajectory_a": ("TENSOR", {"shape_hint": [None, 4, 4]}),
|
||||
# Right operand: Tensor [K, 4, 4] or [4, 4]
|
||||
"trajectory_b": ("TENSOR", {"shape_hint": [None, 4, 4]}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("TENSOR",)
|
||||
RETURN_NAMES = ("trajectory",)
|
||||
FUNCTION = "compose"
|
||||
CATEGORY = "Camera/Pose"
|
||||
DESCRIPTION = "Per-frame matrix product A @ B; a single 4x4 input broadcasts over the other."
|
||||
|
||||
def compose(self, trajectory_a: torch.Tensor, trajectory_b: torch.Tensor) -> Tuple[torch.Tensor]:
|
||||
A = torch.as_tensor(trajectory_a, dtype=torch.float32)
|
||||
B = torch.as_tensor(trajectory_b, dtype=torch.float32)
|
||||
both_single = A.dim() == 2 and B.dim() == 2
|
||||
if A.dim() == 2:
|
||||
A = A.unsqueeze(0)
|
||||
if B.dim() == 2:
|
||||
B = B.unsqueeze(0)
|
||||
if A.dim() != 3 or A.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"Expected trajectory_a of shape [K,4,4] or [4,4], got {tuple(trajectory_a.shape)}")
|
||||
if B.dim() != 3 or B.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"Expected trajectory_b of shape [K,4,4] or [4,4], got {tuple(trajectory_b.shape)}")
|
||||
if A.shape[0] != B.shape[0] and A.shape[0] != 1 and B.shape[0] != 1:
|
||||
raise ValueError(
|
||||
f"Trajectory lengths do not broadcast: {A.shape[0]} vs {B.shape[0]} "
|
||||
"(they must match, or one must be a single 4x4 matrix)."
|
||||
)
|
||||
out = torch.matmul(A, B) # broadcasts [1,4,4] against [K,4,4]
|
||||
if both_single:
|
||||
out = out.squeeze(0)
|
||||
return (out,)
|
||||
|
||||
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"VideoPoseEstimator": VideoPoseEstimator,
|
||||
"TrajectoryInvert": TrajectoryInvert,
|
||||
"TrajectoryCompose": TrajectoryCompose,
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
[project]
|
||||
name = "camera-comfyui"
|
||||
description = "Custom ComfyUI nodes for camera projections (pinhole/fisheye/equirectangular), depth, point clouds, camera trajectories, and 3D/4D Gaussian splatting — including video-to-4D-world workflows."
|
||||
version = "1.1.0"
|
||||
license = { file = "LICENSE" }
|
||||
# Keep in sync with requirements.txt. GitHub/CUDA-flavored extras (vggt,
|
||||
# gsplat) and the ComfyUI-Flux-Inpainting sibling pack are installed by
|
||||
# install.py, which ComfyUI-Manager runs automatically after install.
|
||||
dependencies = [
|
||||
"transformers==4.50.0",
|
||||
"diffusers==0.33.1",
|
||||
"open3d==0.19.0",
|
||||
"protobuf",
|
||||
"click",
|
||||
"timm",
|
||||
"plyfile",
|
||||
"pillow-heif",
|
||||
"matplotlib",
|
||||
"imageio",
|
||||
"imageio-ffmpeg",
|
||||
]
|
||||
|
||||
[project.urls]
|
||||
Repository = "https://github.com/Alexankharin/camera-comfyUI"
|
||||
|
||||
[tool.comfy]
|
||||
PublisherId = "alexk"
|
||||
DisplayName = "camera-comfyUI"
|
||||
# Force-include the SHARP submodule: its files are a gitlink in the parent repo
|
||||
# (not git-tracked files), so without this the registry archive would ship
|
||||
# without submodules/ml-sharpt and ImageToSplat/VideoToFusedSplats would be
|
||||
# unavailable until users clone it manually.
|
||||
includes = ["submodules/ml-sharpt/"]
|
||||
@@ -42,7 +42,14 @@ def map_grid(
|
||||
output_horizontal_fov = torch.tensor(output_horizontal_fov, device=grid_torch.device).float()
|
||||
|
||||
# Calculate vertical field of view for input and output projections
|
||||
output_vertical_fov = output_horizontal_fov # Assuming square aspect ratio
|
||||
# For equirectangular, use 2:1 aspect ratio (vertical FOV = horizontal FOV / 2)
|
||||
if output_projection == "EQUIRECTANGULAR":
|
||||
output_vertical_fov = output_horizontal_fov / 2.0
|
||||
else:
|
||||
output_vertical_fov = output_horizontal_fov # Assuming square aspect ratio for other projections
|
||||
|
||||
# Calculate input vertical FOV based on output grid aspect ratio
|
||||
# This allows the input's vertical range to adapt to the output dimensions
|
||||
input_vertical_fov = input_horizontal_fov * (grid_torch.shape[0] / grid_torch.shape[1])
|
||||
|
||||
# Normalize the grid for vertical FOV adjustment
|
||||
@@ -222,8 +229,8 @@ class ReprojectImage:
|
||||
)
|
||||
|
||||
grid_y, grid_x = torch.meshgrid(
|
||||
torch.linspace(-1, 1, output_width, device=image_tensor.device),
|
||||
torch.linspace(-1, 1, output_height, device=image_tensor.device),
|
||||
torch.linspace(-1, 1, output_width, device=image_tensor.device),
|
||||
indexing="ij"
|
||||
)
|
||||
grid_init = torch.stack((grid_x, grid_y), dim=-1)
|
||||
|
||||
@@ -2,9 +2,17 @@ transformers==4.50.0
|
||||
diffusers==0.33.1
|
||||
open3d==0.19.0
|
||||
protobuf
|
||||
# Runtime deps of the bundled SHARP submodule (submodules/ml-sharpt), so
|
||||
# ImageToSplat & co. work out of the box. torch/torchvision/scipy/tqdm come
|
||||
# with ComfyUI itself; gsplat and vggt are GitHub/CUDA-flavored and are
|
||||
# handled by install.py (run automatically by ComfyUI-Manager).
|
||||
click
|
||||
timm
|
||||
plyfile
|
||||
pillow-heif
|
||||
matplotlib
|
||||
imageio
|
||||
imageio-ffmpeg
|
||||
# open3d is optional, see pointcloud_nodes.py
|
||||
# Python version: 3.12.x (used by embedded python)
|
||||
# All versions pinned to match embedded environment
|
||||
# If using a different Python, adjust versions accordingly
|
||||
# For full reproducibility, consider using a virtual environment
|
||||
# and pip freeze > requirements.txt
|
||||
|
||||
@@ -7,21 +7,30 @@ from typing import Dict, Any, Tuple
|
||||
from tqdm import tqdm # Added tqdm import
|
||||
|
||||
# Import existing pointcloud nodes and projection definitions
|
||||
from .pointcloud_nodes import DepthToPointCloud, TransformPointCloud, ProjectPointCloud, Projection, PointCloudCleaner
|
||||
from .pointcloud_nodes import DepthToPointCloud, TransformPointCloud, ProjectPointCloud, Projection, PointCloudCleaner, interpolate_se3
|
||||
import folder_paths
|
||||
|
||||
# Ensure video_depth_anything is on path
|
||||
video_depth_path = os.path.join("/root/Video-Depth-Anything", "metric_depth")
|
||||
_here = os.path.dirname(os.path.abspath(__file__))
|
||||
|
||||
# climb up 3 levels: camera-comfyUI → custom_nodes → ComfyUI
|
||||
COMFYUI_ROOT = os.path.abspath(os.path.join(_here, os.pardir, os.pardir))
|
||||
|
||||
# point at metric_depth inside the Video-Depth-Anything clone at the ComfyUI root
|
||||
video_depth_path = os.path.join(COMFYUI_ROOT, "Video-Depth-Anything", "metric_depth")
|
||||
|
||||
# insert at front so it always wins
|
||||
if video_depth_path not in sys.path:
|
||||
sys.path.append(video_depth_path)
|
||||
sys.path.insert(0, video_depth_path)
|
||||
NO_VIDEO_DEPTH_ANYTHING= False
|
||||
try:
|
||||
from video_depth_anything.video_depth import VideoDepthAnything
|
||||
print("video_depth_anything module loaded successfully.")
|
||||
except ImportError:
|
||||
VideoDepthAnything = None
|
||||
print("Warning: video_depth_anything module not found. Ensure it is installed correctly.")
|
||||
print("error: ", sys.exc_info()[1])
|
||||
|
||||
print("✅ video_depth_anything module loaded successfully.")
|
||||
except ImportError as e:
|
||||
NO_VIDEO_DEPTH_ANYTHING = True
|
||||
print(
|
||||
f"❌ Could not load video_depth_anything from {video_depth_path!r}: {e}"
|
||||
)
|
||||
|
||||
class VideoCameraMotionSequence:
|
||||
"""
|
||||
@@ -40,6 +49,8 @@ class VideoCameraMotionSequence:
|
||||
"depth_seq": ("TENSOR", {"shape_hint": [None, None, None]}),
|
||||
# Camera trajectory waypoints: Tensor [K, 4, 4]
|
||||
"trajectory": ("TENSOR", {"shape_hint": [None, 4, 4]}),
|
||||
# Optional mask sequence: Tensor [T, H, W] or [T, H, W, 1]
|
||||
"mask_seq": ("MASK", {"shape_hint": [None, None, None], "optional": True}),
|
||||
# Input projection parameters
|
||||
"input_projection": (Projection.PROJECTIONS, {}),
|
||||
"input_horizontal_fov": ("FLOAT", {"default": 90.0}),
|
||||
@@ -78,32 +89,35 @@ class VideoCameraMotionSequence:
|
||||
point_size: int,
|
||||
voxel_size: float,
|
||||
min_points_per_voxel: int,
|
||||
mask_seq: torch.Tensor = None,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
# frames: [T, H, W, 3]
|
||||
# depth_seq: [T, H, W] or [T, H, W, 1]
|
||||
T, H, W, _ = frames.shape
|
||||
|
||||
# Interpolate trajectory to match T
|
||||
K = trajectory.shape[0]
|
||||
if K < 2:
|
||||
interp_traj = trajectory.expand(T, 4, 4).clone()
|
||||
else:
|
||||
idxs = torch.linspace(0, K - 1, T, device=trajectory.device)
|
||||
lower = idxs.floor().long().clamp(max=K - 2)
|
||||
upper = lower + 1
|
||||
alpha = (idxs - lower.float()).unsqueeze(-1).unsqueeze(-1)
|
||||
traj_lower = trajectory[lower]
|
||||
traj_upper = trajectory[upper]
|
||||
interp_traj = traj_lower * (1 - alpha) + traj_upper * alpha
|
||||
# Interpolate trajectory to match T (SE(3): quaternion SLERP on R, lerp on t)
|
||||
interp_traj = interpolate_se3(trajectory, T)
|
||||
|
||||
out_frames = []
|
||||
out_masks = []
|
||||
out_depths = []
|
||||
|
||||
# Add tqdm progress bar for the sequence
|
||||
for frame, depth, pose in tqdm(zip(frames, depth_seq, interp_traj), total=T, desc="Processing video frames"):
|
||||
# If mask_seq is a single mask [H, W] or [H, W, 1], repeat it for all frames
|
||||
if mask_seq is not None:
|
||||
if mask_seq.dim() == 2 or (mask_seq.dim() == 3 and mask_seq.shape[0] == 1):
|
||||
mask_seq = mask_seq.unsqueeze(0) if mask_seq.dim() == 2 else mask_seq
|
||||
mask_seq = mask_seq.repeat(T, 1, 1, 1) if mask_seq.dim() == 4 else mask_seq.repeat(T, 1, 1)
|
||||
|
||||
for i, (frame, depth, pose) in enumerate(tqdm(zip(frames, depth_seq, interp_traj), total=T, desc="Processing video frames")):
|
||||
if depth.dim() == 3 and depth.shape[-1] == 1:
|
||||
depth = depth.squeeze(-1)
|
||||
# Use mask if provided; must be (re)initialized every iteration
|
||||
mask = None
|
||||
if mask_seq is not None:
|
||||
mask = mask_seq[i]
|
||||
if mask.dim() == 3 and mask.shape[-1] == 1:
|
||||
mask = mask.squeeze(-1)
|
||||
# to pointcloud
|
||||
pc, = DepthToPointCloud().depth_to_pointcloud(
|
||||
image=frame.permute(2, 0, 1),
|
||||
@@ -112,7 +126,7 @@ class VideoCameraMotionSequence:
|
||||
depth_scale=depth_scale,
|
||||
invert_depth=invert_depth,
|
||||
depthmap=depth,
|
||||
mask=None,
|
||||
mask=mask,
|
||||
)
|
||||
# optional cleaning
|
||||
if min_points_per_voxel > 1:
|
||||
@@ -213,6 +227,10 @@ class DepthFramesToVideo:
|
||||
raw_color = raw_u8.unsqueeze(1).repeat(1, 3, 1, 1).permute(0, 2, 3, 1)
|
||||
return raw_color, ds_color # [T, 3, H, W] -> [T, H, W, 3]
|
||||
|
||||
# Cache for loaded VideoDepthAnything models, keyed by (checkpoint, device)
|
||||
_VIDEO_DEPTH_MODEL_CACHE: Dict[Tuple[str, str], Any] = {}
|
||||
|
||||
|
||||
class VideoMetricDepthEstimate:
|
||||
"""
|
||||
Estimates metric depth for a sequence of frames using VideoDepthAnything.
|
||||
@@ -243,23 +261,39 @@ class VideoMetricDepthEstimate:
|
||||
input_size: int,
|
||||
max_fps: int,
|
||||
) -> Tuple[torch.Tensor, float]:
|
||||
if VideoDepthAnything is None:
|
||||
raise ImportError("VideoDepthAnything library not found")
|
||||
if NO_VIDEO_DEPTH_ANYTHING:
|
||||
raise ImportError(
|
||||
f"VideoDepthAnything library not found. Clone "
|
||||
f"https://github.com/DepthAnything/Video-Depth-Anything into {COMFYUI_ROOT!r} "
|
||||
f"(expected module path: {video_depth_path!r})."
|
||||
)
|
||||
device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
|
||||
# if max input<1.5 normalize to 0-255
|
||||
if frames.max() < 1.5:
|
||||
frames = (frames * 255)
|
||||
model = VideoDepthAnything(**{"encoder": "vitl", "features": 256, "out_channels": [256,512,1024,1024]})
|
||||
state = torch.load("/root/ComfyUI/models/checkpoints/{}".format(model_checkpoint), map_location='cpu')
|
||||
model.load_state_dict(state, strict=True)
|
||||
model = model.to(device).eval()
|
||||
cache_key = (model_checkpoint, str(device))
|
||||
model = _VIDEO_DEPTH_MODEL_CACHE.get(cache_key)
|
||||
if model is None:
|
||||
# Same checkpoint directory as computed in INPUT_TYPES
|
||||
model_dir = os.path.join(os.getcwd(), "models", "checkpoints")
|
||||
checkpoint_path = os.path.join(model_dir, model_checkpoint)
|
||||
if not os.path.isfile(checkpoint_path):
|
||||
raise FileNotFoundError(f"Checkpoint not found: {checkpoint_path}")
|
||||
model = VideoDepthAnything(**{"encoder": "vitl", "features": 256, "out_channels": [256,512,1024,1024]})
|
||||
state = torch.load(checkpoint_path, map_location='cpu')
|
||||
model.load_state_dict(state, strict=True)
|
||||
model = model.to(device).eval()
|
||||
_VIDEO_DEPTH_MODEL_CACHE[cache_key] = model
|
||||
np_frames = frames.cpu().numpy().astype(np.uint8)
|
||||
metric_depths, fps = model.infer_video_depth(np_frames, max_fps, input_size=input_size, device=device.type, fp32=False)
|
||||
return (torch.from_numpy(metric_depths), float(fps))
|
||||
|
||||
# Register nodes
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"VideoCameraMotionSequence": VideoCameraMotionSequence,
|
||||
"VideoMetricDepthEstimate": VideoMetricDepthEstimate,
|
||||
"DepthFramesToVideo": DepthFramesToVideo,
|
||||
}
|
||||
if NO_VIDEO_DEPTH_ANYTHING:
|
||||
NODE_CLASS_MAPPINGS = {}
|
||||
else:
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"VideoCameraMotionSequence": VideoCameraMotionSequence,
|
||||
"VideoMetricDepthEstimate": VideoMetricDepthEstimate,
|
||||
"DepthFramesToVideo": DepthFramesToVideo,
|
||||
}
|
||||
|
||||
|
After Width: | Height: | Size: 40 KiB |
|
After Width: | Height: | Size: 27 KiB |
@@ -1 +1,289 @@
|
||||
{"id":"8c6e5ec1-4ff2-42d9-9408-fcd0a23d362a","revision":0,"last_node_id":11,"last_link_id":17,"nodes":[{"id":3,"type":"PreviewImage","pos":[-5.768195629119873,204.72561645507812],"size":[210,246],"flags":{},"order":2,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":16}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":4,"type":"MaskToImage","pos":[-30.901016235351562,532.864013671875],"size":[264.5999755859375,26],"flags":{},"order":3,"mode":0,"inputs":[{"localized_name":"mask","name":"mask","type":"MASK","link":17}],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[4]}],"properties":{"Node name for S&R":"MaskToImage"},"widgets_values":[]},{"id":2,"type":"LoadImage","pos":[-843.689208984375,354.9311828613281],"size":[315,314],"flags":{},"order":0,"mode":0,"inputs":[],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[15]},{"localized_name":"MASK","name":"MASK","type":"MASK","links":[]}],"properties":{"Node name for S&R":"LoadImage"},"widgets_values":["flux_dev_example.png","image",""]},{"id":5,"type":"PreviewImage","pos":[296.6977844238281,611.6702880859375],"size":[210,246],"flags":{},"order":4,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":4}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":11,"type":"OutpaintAnyProjection","pos":[-498.9301452636719,317.144287109375],"size":[400,492],"flags":{},"order":1,"mode":0,"inputs":[{"localized_name":"image","name":"image","type":"IMAGE","link":15},{"localized_name":"mask","name":"mask","shape":7,"type":"MASK","link":null}],"outputs":[{"localized_name":"final_image","name":"final_image","type":"IMAGE","links":[16]},{"localized_name":"needs_inpaint_mask","name":"needs_inpaint_mask","type":"MASK","links":[17]}],"properties":{"Node name for S&R":"OutpaintAnyProjection"},"widgets_values":["PINHOLE",90,"FISHEYE",180,4096,4096,"PINHOLE",90,1024,45,0,"",10,false,30,1,false]}],"links":[[4,4,0,5,0,"IMAGE"],[15,2,0,11,0,"IMAGE"],[16,11,0,3,0,"IMAGE"],[17,11,1,4,0,"MASK"]],"groups":[],"config":{},"extra":{"ds":{"scale":0.8390545288824094,"offset":[703.1897327574803,-308.43624510132975]}},"version":0.4}
|
||||
{
|
||||
"id": "8c6e5ec1-4ff2-42d9-9408-fcd0a23d362a",
|
||||
"revision": 0,
|
||||
"last_node_id": 12,
|
||||
"last_link_id": 17,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 3,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
-5.768195629119873,
|
||||
204.72561645507812
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 16
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
],
|
||||
"title": "Result"
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "MaskToImage",
|
||||
"pos": [
|
||||
-30.901016235351562,
|
||||
532.864013671875
|
||||
],
|
||||
"size": [
|
||||
264.5999755859375,
|
||||
26
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"type": "MASK",
|
||||
"link": 17
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
4
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "MaskToImage"
|
||||
},
|
||||
"widgets_values": []
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "LoadImage",
|
||||
"pos": [
|
||||
-843.689208984375,
|
||||
354.9311828613281
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
15
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": []
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"camera_example_pinhole.jpg",
|
||||
"image",
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
296.6977844238281,
|
||||
611.6702880859375
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 4
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
],
|
||||
"title": "Holes still to fill"
|
||||
},
|
||||
{
|
||||
"id": 11,
|
||||
"type": "OutpaintAnyProjection",
|
||||
"pos": [
|
||||
-498.9301452636719,
|
||||
317.144287109375
|
||||
],
|
||||
"size": [
|
||||
400,
|
||||
492
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 15
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"shape": 7,
|
||||
"type": "MASK",
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "final_image",
|
||||
"name": "final_image",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
16
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "needs_inpaint_mask",
|
||||
"name": "needs_inpaint_mask",
|
||||
"type": "MASK",
|
||||
"links": [
|
||||
17
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "OutpaintAnyProjection"
|
||||
},
|
||||
"widgets_values": [
|
||||
"PINHOLE",
|
||||
90,
|
||||
"FISHEYE",
|
||||
180,
|
||||
4096,
|
||||
4096,
|
||||
"PINHOLE",
|
||||
90,
|
||||
1024,
|
||||
45,
|
||||
0,
|
||||
"",
|
||||
10,
|
||||
false,
|
||||
30,
|
||||
1,
|
||||
false
|
||||
],
|
||||
"title": "Outpaint one patch (yaw +45°)"
|
||||
},
|
||||
{
|
||||
"id": 12,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
-1403.689208984375,
|
||||
204.72561645507812
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
288
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# OutpaintAnyProjection smoke test\n\nSingle-node sanity check: a 90° pinhole image is placed on a 180° fisheye canvas and one 90° pinhole patch at yaw +45° is Flux-inpainted (10 steps for speed).\n\nOutputs: the partially outpainted canvas and the *remaining holes* mask — chain more OutpaintAnyProjection nodes at other angles to fill it (see `Outpaint_fisheye180.json`).\n\n- **Set:** input image; prompt inside the node (empty = unconditional).\n- **Requires:** `custom_nodes/inpainting_flux` (installed automatically by this pack's install.py) — downloads FLUX.1-Fill NF4 weights on first run."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
4,
|
||||
4,
|
||||
0,
|
||||
5,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
15,
|
||||
2,
|
||||
0,
|
||||
11,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
16,
|
||||
11,
|
||||
0,
|
||||
3,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
17,
|
||||
11,
|
||||
1,
|
||||
4,
|
||||
0,
|
||||
"MASK"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.8390545288824094,
|
||||
"offset": [
|
||||
703.1897327574803,
|
||||
-308.43624510132975
|
||||
]
|
||||
},
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"id": "312a3a27-6189-4cd7-8c9f-5a62881c623e",
|
||||
"revision": 0,
|
||||
"last_node_id": 18,
|
||||
"last_node_id": 19,
|
||||
"last_link_id": 37,
|
||||
"nodes": [
|
||||
{
|
||||
@@ -38,7 +38,7 @@
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Fisheye_outpainted_flux_dev.png",
|
||||
"camera_example_fisheye.jpg",
|
||||
"image"
|
||||
]
|
||||
},
|
||||
@@ -213,7 +213,9 @@
|
||||
90,
|
||||
1024,
|
||||
4096,
|
||||
21
|
||||
"SOFTMERGE",
|
||||
21,
|
||||
1
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -326,14 +328,11 @@
|
||||
10,
|
||||
7.5,
|
||||
5,
|
||||
"PINHOLE",
|
||||
90,
|
||||
1024,
|
||||
1024,
|
||||
"open",
|
||||
9,
|
||||
15
|
||||
]
|
||||
0.07,
|
||||
3,
|
||||
"Depth-Anything-V2-Metric-Indoor-Base-hf"
|
||||
],
|
||||
"title": "Enrich cloud along trajectory (Flux outpaint)"
|
||||
},
|
||||
{
|
||||
"id": 18,
|
||||
@@ -411,8 +410,36 @@
|
||||
90,
|
||||
1024,
|
||||
1024,
|
||||
2
|
||||
]
|
||||
2,
|
||||
0,
|
||||
false,
|
||||
false
|
||||
],
|
||||
"title": "Orbit preview of enriched cloud"
|
||||
},
|
||||
{
|
||||
"id": 19,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
982.239013671875,
|
||||
1182.49755859375
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Point cloud enricher (outpaint along a trajectory)\n\nFisheye image → metric depth → point cloud, then **PointcloudTrajectoryEnricher** walks the loaded camera trajectory: at each pose it renders the cloud, Flux-inpaints the disocclusion holes, re-estimates depth, aligns it and merges the new points into the cloud. The enriched cloud is saved and previewed as an orbit video.\n\nThis is the one-node version of `pointcloud_inpaint.json`.\n\n- **Set:** input image; trajectory .npy (record one with SaveTrajectory); prompt inside the enricher.\n- **Requires:** inpainting_flux (auto-installed), Depth-Anything V2, FLUX.1-Fill NF4 weights.\n- **Note:** 2026-07 schema migration — the enricher's render/reproject options are now internal; voxel merge params were reset to defaults."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
@@ -513,7 +540,47 @@
|
||||
"TENSOR"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"title": "1. Fisheye → point cloud",
|
||||
"bounding": [
|
||||
1522.239013671875,
|
||||
1122.49755859375,
|
||||
1332.4192962646484,
|
||||
589.1494140625
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"title": "2. Enrich along trajectory",
|
||||
"bounding": [
|
||||
1964.2667236328125,
|
||||
1479.796630859375,
|
||||
1073.373291015625,
|
||||
614.0
|
||||
],
|
||||
"color": "#a1309b",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"title": "3. Save & preview",
|
||||
"bounding": [
|
||||
2283.51123046875,
|
||||
1545.8380126953125,
|
||||
1130.08935546875,
|
||||
726.26318359375
|
||||
],
|
||||
"color": "#8A8",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
}
|
||||
],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
@@ -523,7 +590,8 @@
|
||||
-1433.3959538925521
|
||||
]
|
||||
},
|
||||
"frontendVersion": "1.19.9"
|
||||
"frontendVersion": "1.19.9",
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
}
|
||||
|
||||
|
After Width: | Height: | Size: 65 KiB |
@@ -1,2 +0,0 @@
|
||||
# This file marks the workflows directory as a Python package.
|
||||
NODE_CLASS_MAPPINGS={}
|
||||
|
After Width: | Height: | Size: 38 KiB |
@@ -1 +1,331 @@
|
||||
{"id":"de69fe58-2f01-4527-92b1-681a151b1ce3","revision":0,"last_node_id":7,"last_link_id":8,"nodes":[{"id":3,"type":"LoadImage","pos":[-897.7078247070312,314.2335205078125],"size":[315,314],"flags":{},"order":0,"mode":0,"inputs":[],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[2]},{"localized_name":"MASK","name":"MASK","type":"MASK","links":null}],"properties":{"Node name for S&R":"LoadImage"},"widgets_values":["example.png","image",""]},{"id":1,"type":"TransformToMatrix","pos":[-871.939697265625,694.5091552734375],"size":[315,154],"flags":{},"order":1,"mode":0,"inputs":[],"outputs":[{"localized_name":"MAT_4X4","name":"MAT_4X4","type":"MAT_4X4","links":[5]}],"properties":{"Node name for S&R":"TransformToMatrix"},"widgets_values":[0,0,0,60,0]},{"id":6,"type":"MaskToImage","pos":[-169.98599243164062,763.4010009765625],"size":[264.5999755859375,26],"flags":{},"order":4,"mode":0,"inputs":[{"localized_name":"mask","name":"mask","type":"MASK","link":6}],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[7]}],"properties":{"Node name for S&R":"MaskToImage"}},{"id":4,"type":"PreviewImage","pos":[299.262451171875,609.4740600585938],"size":[210,246],"flags":{},"order":5,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":7}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":7,"type":"PreviewImage","pos":[-14.518107414245605,366.5712585449219],"size":[210,246],"flags":{},"order":3,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":8}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":2,"type":"ReprojectImage","pos":[-468.4571838378906,358.2191467285156],"size":[315,266],"flags":{},"order":2,"mode":0,"inputs":[{"localized_name":"image","name":"image","type":"IMAGE","link":2},{"localized_name":"mask","name":"mask","shape":7,"type":"MASK","link":null},{"localized_name":"transform_matrix","name":"transform_matrix","shape":7,"type":"MAT_4X4","link":5}],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[8]},{"localized_name":"MASK","name":"MASK","type":"MASK","links":[6]}],"properties":{"Node name for S&R":"ReprojectImage"},"widgets_values":[90,180,"PINHOLE","EQUIRECTANGULAR",1024,1024,false,7]}],"links":[[2,3,0,2,0,"IMAGE"],[5,1,0,2,2,"MAT_4X4"],[6,2,1,6,0,"MASK"],[7,6,0,4,0,"IMAGE"],[8,2,0,7,0,"IMAGE"]],"groups":[],"config":{},"extra":{"ds":{"scale":0.8264462809917362,"offset":[806.6857406850668,-309.49182798364893]}},"version":0.4}
|
||||
{
|
||||
"id": "de69fe58-2f01-4527-92b1-681a151b1ce3",
|
||||
"revision": 0,
|
||||
"last_node_id": 8,
|
||||
"last_link_id": 8,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 3,
|
||||
"type": "LoadImage",
|
||||
"pos": [
|
||||
-897.7078247070312,
|
||||
314.2335205078125
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
2
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"camera_example_pinhole.jpg",
|
||||
"image",
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 1,
|
||||
"type": "TransformToMatrix",
|
||||
"pos": [
|
||||
-871.939697265625,
|
||||
694.5091552734375
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "MAT_4X4",
|
||||
"name": "MAT_4X4",
|
||||
"type": "MAT_4X4",
|
||||
"links": [
|
||||
5
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformToMatrix"
|
||||
},
|
||||
"widgets_values": [
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
60,
|
||||
0
|
||||
],
|
||||
"title": "Rotate camera (pitch 60°)"
|
||||
},
|
||||
{
|
||||
"id": 6,
|
||||
"type": "MaskToImage",
|
||||
"pos": [
|
||||
-169.98599243164062,
|
||||
763.4010009765625
|
||||
],
|
||||
"size": [
|
||||
264.5999755859375,
|
||||
26
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"type": "MASK",
|
||||
"link": 6
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
7
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "MaskToImage"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
299.262451171875,
|
||||
609.4740600585938
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 7
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
],
|
||||
"title": "Coverage mask (white = hole)"
|
||||
},
|
||||
{
|
||||
"id": 7,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
-14.518107414245605,
|
||||
366.5712585449219
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 8
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
],
|
||||
"title": "Reprojected image"
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "ReprojectImage",
|
||||
"pos": [
|
||||
-468.4571838378906,
|
||||
358.2191467285156
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
266
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 2
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"shape": 7,
|
||||
"type": "MASK",
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "transform_matrix",
|
||||
"name": "transform_matrix",
|
||||
"shape": 7,
|
||||
"type": "MAT_4X4",
|
||||
"link": 5
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
8
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": [
|
||||
6
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "ReprojectImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
90,
|
||||
180,
|
||||
"PINHOLE",
|
||||
"EQUIRECTANGULAR",
|
||||
1024,
|
||||
1024,
|
||||
false,
|
||||
7
|
||||
],
|
||||
"title": "Pinhole 90° → Equirect 180°"
|
||||
},
|
||||
{
|
||||
"id": 8,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
-1457.7078247070312,
|
||||
314.2335205078125
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Camera reprojection demo\n\nMinimal example of the two core nodes: **TransformToMatrix** rotates the virtual camera (theta = 60° pitch) and **ReprojectImage** converts a 90° pinhole image into a 180° equirectangular view.\n\nPreviews show the reprojected image and the coverage mask (white = pixels the source image cannot see).\n\n- **Set:** the image in LoadImage.\n- **Requires:** nothing beyond this pack (no models).\n- **Try:** switch `output_projection` to FISHEYE, or raise `feathering` to soften the mask edge."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
2,
|
||||
3,
|
||||
0,
|
||||
2,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
5,
|
||||
1,
|
||||
0,
|
||||
2,
|
||||
2,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
6,
|
||||
2,
|
||||
1,
|
||||
6,
|
||||
0,
|
||||
"MASK"
|
||||
],
|
||||
[
|
||||
7,
|
||||
6,
|
||||
0,
|
||||
4,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
8,
|
||||
2,
|
||||
0,
|
||||
7,
|
||||
0,
|
||||
"IMAGE"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.8264462809917362,
|
||||
"offset": [
|
||||
806.6857406850668,
|
||||
-309.49182798364893
|
||||
]
|
||||
},
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
|
||||
|
After Width: | Height: | Size: 16 KiB |
@@ -1 +1,469 @@
|
||||
{"id":"8e48e6c6-735b-4131-9b4d-5847d2948c18","revision":0,"last_node_id":161,"last_link_id":330,"nodes":[{"id":156,"type":"DepthToImageNode","pos":[101.64265441894531,607.9051513671875],"size":[315,58],"flags":{},"order":2,"mode":0,"inputs":[{"localized_name":"depth","name":"depth","type":"TENSOR","link":323}],"outputs":[{"localized_name":"depth image","name":"depth image","type":"IMAGE","links":[322]}],"properties":{"Node name for S&R":"DepthToImageNode"},"widgets_values":[false]},{"id":157,"type":"PreviewImage","pos":[449.5372314453125,592.8480224609375],"size":[210,246],"flags":{},"order":5,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":322}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":159,"type":"PreviewImage","pos":[128.622802734375,835.9542236328125],"size":[210,246],"flags":{},"order":6,"mode":0,"inputs":[{"localized_name":"images","name":"images","type":"IMAGE","link":326}],"outputs":[],"properties":{"Node name for S&R":"PreviewImage"},"widgets_values":[""]},{"id":158,"type":"MaskToImage","pos":[105.04651641845703,715.2989501953125],"size":[264.5999755859375,26],"flags":{},"order":3,"mode":0,"inputs":[{"localized_name":"mask","name":"mask","type":"MASK","link":325}],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[326]}],"properties":{"Node name for S&R":"MaskToImage"}},{"id":154,"type":"LoadImage","pos":[-601.5926513671875,610.8336181640625],"size":[315,314],"flags":{},"order":0,"mode":0,"inputs":[],"outputs":[{"localized_name":"IMAGE","name":"IMAGE","type":"IMAGE","links":[324,327]},{"localized_name":"MASK","name":"MASK","type":"MASK","links":null}],"properties":{"Node name for S&R":"LoadImage"},"widgets_values":["Saved_fisheye_00001_.png","image",""]},{"id":155,"type":"FisheyeDepthEstimator","pos":[-258.0684814453125,592.8478393554688],"size":[315,198],"flags":{},"order":1,"mode":0,"inputs":[{"localized_name":"image","name":"image","type":"IMAGE","link":324}],"outputs":[{"localized_name":"depthmap","name":"depthmap","type":"TENSOR","links":[323,329]},{"localized_name":"mask","name":"mask","type":"MASK","links":[325,328]}],"properties":{"Node name for S&R":"FisheyeDepthEstimator"},"widgets_values":["Depth-Anything-V2-Metric-Indoor-Base-hf",1,90,1024,4096,25]},{"id":160,"type":"DepthToPointCloud","pos":[398.363525390625,895.5885620117188],"size":[315,170],"flags":{},"order":4,"mode":0,"inputs":[{"localized_name":"image","name":"image","type":"IMAGE","link":327},{"localized_name":"depthmap","name":"depthmap","shape":7,"type":"TENSOR","link":329},{"localized_name":"mask","name":"mask","shape":7,"type":"MASK","link":328}],"outputs":[{"localized_name":"pointcloud","name":"pointcloud","type":"TENSOR","links":[330]}],"properties":{"Node name for S&R":"DepthToPointCloud"},"widgets_values":["FISHEYE",180,1,false]},{"id":161,"type":"SavePointCloud","pos":[785.9862670898438,881.0266723632812],"size":[315,82],"flags":{},"order":7,"mode":0,"inputs":[{"localized_name":"pointcloud","name":"pointcloud","type":"TENSOR","link":330}],"outputs":[],"properties":{"Node name for S&R":"SavePointCloud"},"widgets_values":["kitchen","npy"]}],"links":[[322,156,0,157,0,"IMAGE"],[323,155,0,156,0,"TENSOR"],[324,154,0,155,0,"IMAGE"],[325,155,1,158,0,"MASK"],[326,158,0,159,0,"IMAGE"],[327,154,0,160,0,"IMAGE"],[328,155,1,160,2,"MASK"],[329,155,0,160,1,"TENSOR"],[330,160,0,161,0,"TENSOR"]],"groups":[],"config":{},"extra":{"ds":{"scale":0.9229599817706441,"offset":[115.11465258583665,-662.5990021779805]},"frontendVersion":"1.18.9"},"version":0.4}
|
||||
{
|
||||
"id": "8e48e6c6-735b-4131-9b4d-5847d2948c18",
|
||||
"revision": 0,
|
||||
"last_node_id": 162,
|
||||
"last_link_id": 330,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 156,
|
||||
"type": "DepthToImageNode",
|
||||
"pos": [
|
||||
101.64265441894531,
|
||||
607.9051513671875
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
58
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "depth",
|
||||
"name": "depth",
|
||||
"type": "TENSOR",
|
||||
"link": 323
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "depth image",
|
||||
"name": "depth image",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
322
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "DepthToImageNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
false
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 157,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
449.5372314453125,
|
||||
592.8480224609375
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 322
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 159,
|
||||
"type": "PreviewImage",
|
||||
"pos": [
|
||||
128.622802734375,
|
||||
835.9542236328125
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
246
|
||||
],
|
||||
"flags": {},
|
||||
"order": 6,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 326
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "PreviewImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 158,
|
||||
"type": "MaskToImage",
|
||||
"pos": [
|
||||
105.04651641845703,
|
||||
715.2989501953125
|
||||
],
|
||||
"size": [
|
||||
264.5999755859375,
|
||||
26
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"type": "MASK",
|
||||
"link": 325
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
326
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "MaskToImage"
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 154,
|
||||
"type": "LoadImage",
|
||||
"pos": [
|
||||
-601.5926513671875,
|
||||
610.8336181640625
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
324,
|
||||
327
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"camera_example_fisheye.jpg",
|
||||
"image",
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 155,
|
||||
"type": "FisheyeDepthEstimator",
|
||||
"pos": [
|
||||
-258.0684814453125,
|
||||
592.8478393554688
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
198
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 324
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "depthmap",
|
||||
"name": "depthmap",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
323,
|
||||
329
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"type": "MASK",
|
||||
"links": [
|
||||
325,
|
||||
328
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "FisheyeDepthEstimator"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Depth-Anything-V2-Metric-Indoor-Base-hf",
|
||||
1,
|
||||
90,
|
||||
1024,
|
||||
4096,
|
||||
"SOFTMERGE",
|
||||
25,
|
||||
1
|
||||
],
|
||||
"title": "Metric depth (fisheye-aware)"
|
||||
},
|
||||
{
|
||||
"id": 160,
|
||||
"type": "DepthToPointCloud",
|
||||
"pos": [
|
||||
398.363525390625,
|
||||
895.5885620117188
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
170
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 327
|
||||
},
|
||||
{
|
||||
"localized_name": "depthmap",
|
||||
"name": "depthmap",
|
||||
"shape": 7,
|
||||
"type": "TENSOR",
|
||||
"link": 329
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"shape": 7,
|
||||
"type": "MASK",
|
||||
"link": 328
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
330
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "DepthToPointCloud"
|
||||
},
|
||||
"widgets_values": [
|
||||
"FISHEYE",
|
||||
180,
|
||||
1,
|
||||
false
|
||||
],
|
||||
"title": "Depth → point cloud"
|
||||
},
|
||||
{
|
||||
"id": 161,
|
||||
"type": "SavePointCloud",
|
||||
"pos": [
|
||||
785.9862670898438,
|
||||
881.0266723632812
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
82
|
||||
],
|
||||
"flags": {},
|
||||
"order": 7,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"link": 330
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "SavePointCloud"
|
||||
},
|
||||
"widgets_values": [
|
||||
"kitchen",
|
||||
"npy"
|
||||
],
|
||||
"title": "Save .ply / .npy"
|
||||
},
|
||||
{
|
||||
"id": 162,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
-1161.5926513671875,
|
||||
592.8478393554688
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Fisheye 180° → point cloud\n\nEstimates metric depth directly on a 180° fisheye image — **FisheyeDepthEstimator** internally splits it into pinhole views, runs Depth-Anything V2 on each and merges the depths back (SOFTMERGE) — then unprojects image + depth to a 3D point cloud and saves it.\n\nPreviews: colorized depth and the validity mask.\n\n- **Set:** fisheye input image (e.g. produced by `Outpaint_fisheye180.json`); filename in SavePointCloud.\n- **Requires:** Depth-Anything V2 (auto-downloads from HuggingFace).\n- **Next:** view the cloud with `test_pointcloud_loading.json` or synthesize a second eye with `sbs180_workflow.json`."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
322,
|
||||
156,
|
||||
0,
|
||||
157,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
323,
|
||||
155,
|
||||
0,
|
||||
156,
|
||||
0,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
324,
|
||||
154,
|
||||
0,
|
||||
155,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
325,
|
||||
155,
|
||||
1,
|
||||
158,
|
||||
0,
|
||||
"MASK"
|
||||
],
|
||||
[
|
||||
326,
|
||||
158,
|
||||
0,
|
||||
159,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
327,
|
||||
154,
|
||||
0,
|
||||
160,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
328,
|
||||
155,
|
||||
1,
|
||||
160,
|
||||
2,
|
||||
"MASK"
|
||||
],
|
||||
[
|
||||
329,
|
||||
155,
|
||||
0,
|
||||
160,
|
||||
1,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
330,
|
||||
160,
|
||||
0,
|
||||
161,
|
||||
0,
|
||||
"TENSOR"
|
||||
]
|
||||
],
|
||||
"groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"title": "1. Fisheye metric depth",
|
||||
"bounding": [
|
||||
-621.5926513671875,
|
||||
532.8478393554688,
|
||||
1301.1298828125,
|
||||
569.1063842773438
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"title": "2. Unproject & save",
|
||||
"bounding": [
|
||||
378.363525390625,
|
||||
821.0266723632812,
|
||||
742.6227416992188,
|
||||
264.5618896484375
|
||||
],
|
||||
"color": "#a1309b",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
}
|
||||
],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.9229599817706441,
|
||||
"offset": [
|
||||
115.11465258583665,
|
||||
-662.5990021779805
|
||||
]
|
||||
},
|
||||
"frontendVersion": "1.18.9",
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
|
||||
|
After Width: | Height: | Size: 38 KiB |
|
After Width: | Height: | Size: 50 KiB |
@@ -0,0 +1,781 @@
|
||||
{
|
||||
"id": "016c13e5-c6a7-4b2e-adfc-18e9e969f21e",
|
||||
"revision": 0,
|
||||
"last_node_id": 21,
|
||||
"last_link_id": 32,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 10,
|
||||
"type": "SaveWEBM",
|
||||
"pos": [
|
||||
3352.103515625,
|
||||
1535.6951904296875
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
437
|
||||
],
|
||||
"flags": {},
|
||||
"order": 8,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 24
|
||||
},
|
||||
{
|
||||
"localized_name": "filename_prefix",
|
||||
"name": "filename_prefix",
|
||||
"type": "STRING",
|
||||
"widget": {
|
||||
"name": "filename_prefix"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "codec",
|
||||
"name": "codec",
|
||||
"type": "COMBO",
|
||||
"widget": {
|
||||
"name": "codec"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "fps",
|
||||
"name": "fps",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "fps"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "crf",
|
||||
"name": "crf",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "crf"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"ComfyUI",
|
||||
"vp9",
|
||||
24,
|
||||
32
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "TransformPointCloud",
|
||||
"pos": [
|
||||
2458.775390625,
|
||||
1405.860595703125
|
||||
],
|
||||
"size": [
|
||||
329.20001220703125,
|
||||
46
|
||||
],
|
||||
"flags": {},
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"link": 4
|
||||
},
|
||||
{
|
||||
"localized_name": "transform_matrix",
|
||||
"name": "transform_matrix",
|
||||
"type": "MAT_4X4",
|
||||
"link": 5
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "transformed pointcloud",
|
||||
"name": "transformed pointcloud",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
23
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformPointCloud"
|
||||
},
|
||||
"widgets_values": []
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "TransformToMatrix",
|
||||
"pos": [
|
||||
2296.978759765625,
|
||||
1742.214599609375
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "shiftX",
|
||||
"name": "shiftX",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "shiftX"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "shiftY",
|
||||
"name": "shiftY",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "shiftY"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "shiftZ",
|
||||
"name": "shiftZ",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "shiftZ"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "theta",
|
||||
"name": "theta",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "theta"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "phi",
|
||||
"name": "phi",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "phi"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "transformation matrix",
|
||||
"name": "transformation matrix",
|
||||
"type": "MAT_4X4",
|
||||
"links": [
|
||||
5,
|
||||
25
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformToMatrix"
|
||||
},
|
||||
"widgets_values": [
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0
|
||||
],
|
||||
"title": "Camera move (edit me)"
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "DepthToPointCloud",
|
||||
"pos": [
|
||||
2099.81640625,
|
||||
1385.9761962890625
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
170
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 1
|
||||
},
|
||||
{
|
||||
"localized_name": "depthmap",
|
||||
"name": "depthmap",
|
||||
"shape": 7,
|
||||
"type": "TENSOR",
|
||||
"link": 32
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"shape": 7,
|
||||
"type": "MASK",
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "input_projection",
|
||||
"name": "input_projection",
|
||||
"type": "COMBO",
|
||||
"widget": {
|
||||
"name": "input_projection"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "input_horizontal_fov",
|
||||
"name": "input_horizontal_fov",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "input_horizontal_fov"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "depth_scale",
|
||||
"name": "depth_scale",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "depth_scale"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "invert_depth",
|
||||
"name": "invert_depth",
|
||||
"type": "BOOLEAN",
|
||||
"widget": {
|
||||
"name": "invert_depth"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
4,
|
||||
26
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "DepthToPointCloud"
|
||||
},
|
||||
"widgets_values": [
|
||||
"PINHOLE",
|
||||
60,
|
||||
1,
|
||||
false
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 14,
|
||||
"type": "CameraMotionNode",
|
||||
"pos": [
|
||||
2984.30224609375,
|
||||
1486.6483154296875
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
198
|
||||
],
|
||||
"flags": {},
|
||||
"order": 7,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"link": 23
|
||||
},
|
||||
{
|
||||
"localized_name": "trajectory",
|
||||
"name": "trajectory",
|
||||
"type": "TENSOR",
|
||||
"link": 27
|
||||
},
|
||||
{
|
||||
"localized_name": "n_points",
|
||||
"name": "n_points",
|
||||
"type": "INT",
|
||||
"widget": {
|
||||
"name": "n_points"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "output_projection",
|
||||
"name": "output_projection",
|
||||
"type": "COMBO",
|
||||
"widget": {
|
||||
"name": "output_projection"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "output_horizontal_fov",
|
||||
"name": "output_horizontal_fov",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "output_horizontal_fov"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "output_width",
|
||||
"name": "output_width",
|
||||
"type": "INT",
|
||||
"widget": {
|
||||
"name": "output_width"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "output_height",
|
||||
"name": "output_height",
|
||||
"type": "INT",
|
||||
"widget": {
|
||||
"name": "output_height"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "point_size",
|
||||
"name": "point_size",
|
||||
"type": "INT",
|
||||
"widget": {
|
||||
"name": "point_size"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "motion_frames",
|
||||
"name": "motion_frames",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
24
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "CameraMotionNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
10,
|
||||
"PINHOLE",
|
||||
60,
|
||||
1024,
|
||||
1024,
|
||||
2,
|
||||
0,
|
||||
false,
|
||||
false
|
||||
],
|
||||
"title": "Render fly-through"
|
||||
},
|
||||
{
|
||||
"id": 17,
|
||||
"type": "CameraTrajectoryNode",
|
||||
"pos": [
|
||||
2722.265380859375,
|
||||
1772.097412109375
|
||||
],
|
||||
"size": [
|
||||
317.4000244140625,
|
||||
46
|
||||
],
|
||||
"flags": {},
|
||||
"order": 6,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "pointcloud",
|
||||
"name": "pointcloud",
|
||||
"type": "TENSOR",
|
||||
"link": 26
|
||||
},
|
||||
{
|
||||
"localized_name": "initial_matrix",
|
||||
"name": "initial_matrix",
|
||||
"type": "MAT_4X4",
|
||||
"link": 25
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "trajectory",
|
||||
"name": "trajectory",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
27
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "CameraTrajectoryNode"
|
||||
},
|
||||
"widgets_values": [],
|
||||
"title": "Build trajectory"
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"type": "DepthEstimatorNode",
|
||||
"pos": [
|
||||
1570.1402587890625,
|
||||
1859.879638671875
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
82
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 3
|
||||
},
|
||||
{
|
||||
"localized_name": "model_name",
|
||||
"name": "model_name",
|
||||
"type": "STRING",
|
||||
"widget": {
|
||||
"name": "model_name"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "depth_scale",
|
||||
"name": "depth_scale",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "depth_scale"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "depth tensor",
|
||||
"name": "depth tensor",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
31
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "DepthEstimatorNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Depth-Anything-V2-Metric-Indoor-Base-hf",
|
||||
1.0000000000000002,
|
||||
1
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 20,
|
||||
"type": "ZDepthToRayDepthNode",
|
||||
"pos": [
|
||||
1969.0958251953125,
|
||||
1782.2877197265625
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
58
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "depth",
|
||||
"name": "depth",
|
||||
"type": "TENSOR",
|
||||
"link": 31
|
||||
},
|
||||
{
|
||||
"localized_name": "fov",
|
||||
"name": "fov",
|
||||
"type": "FLOAT",
|
||||
"widget": {
|
||||
"name": "fov"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "ray depth",
|
||||
"name": "ray depth",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
32
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "ZDepthToRayDepthNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
90
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 1,
|
||||
"type": "LoadImage",
|
||||
"pos": [
|
||||
1575.78955078125,
|
||||
1419.3006591796875
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "COMBO",
|
||||
"widget": {
|
||||
"name": "image"
|
||||
},
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "choose file to upload",
|
||||
"name": "upload",
|
||||
"type": "IMAGEUPLOAD",
|
||||
"widget": {
|
||||
"name": "upload"
|
||||
},
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
1,
|
||||
3
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"camera_example_pinhole.jpg",
|
||||
"image"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 21,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
1010.1402587890625,
|
||||
1385.9761962890625
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
262
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Point cloud walker (fly-through video)\n\nSingle image → metric depth → point cloud, then **CameraTrajectoryNode** derives a camera path and **CameraMotionNode** renders a fly-through saved as WEBM.\n\n- **Set:** input image; the move in TransformToMatrix (shift XYZ + theta/phi); frames-per-segment (`n_points`) in CameraMotionNode.\n- **Requires:** Depth-Anything V2 (auto-download).\n- **Note:** the pinhole FOV here is 60° — keep DepthToPointCloud and CameraMotionNode FOVs consistent with your input."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
1,
|
||||
1,
|
||||
0,
|
||||
2,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
3,
|
||||
1,
|
||||
0,
|
||||
3,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
4,
|
||||
2,
|
||||
0,
|
||||
4,
|
||||
0,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
5,
|
||||
5,
|
||||
0,
|
||||
4,
|
||||
1,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
23,
|
||||
4,
|
||||
0,
|
||||
14,
|
||||
0,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
24,
|
||||
14,
|
||||
0,
|
||||
10,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
25,
|
||||
5,
|
||||
0,
|
||||
17,
|
||||
1,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
26,
|
||||
2,
|
||||
0,
|
||||
17,
|
||||
0,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
27,
|
||||
17,
|
||||
0,
|
||||
14,
|
||||
1,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
31,
|
||||
3,
|
||||
0,
|
||||
20,
|
||||
0,
|
||||
"TENSOR"
|
||||
],
|
||||
[
|
||||
32,
|
||||
20,
|
||||
0,
|
||||
2,
|
||||
1,
|
||||
"TENSOR"
|
||||
]
|
||||
],
|
||||
"groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"title": "1. Image → point cloud",
|
||||
"bounding": [
|
||||
1550.1402587890625,
|
||||
1325.9761962890625,
|
||||
884.6761474609375,
|
||||
635.9034423828125
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"title": "2. Trajectory",
|
||||
"bounding": [
|
||||
2276.978759765625,
|
||||
1345.860595703125,
|
||||
782.6866455078125,
|
||||
570.35400390625
|
||||
],
|
||||
"color": "#a1309b",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"title": "3. Render & save",
|
||||
"bounding": [
|
||||
2964.30224609375,
|
||||
1426.6483154296875,
|
||||
722.80126953125,
|
||||
566.046875
|
||||
],
|
||||
"color": "#8A8",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
}
|
||||
],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 1.083470594338842,
|
||||
"offset": [
|
||||
-1328.4875281402603,
|
||||
-1380.0248413907814
|
||||
]
|
||||
},
|
||||
"frontendVersion": "1.18.9",
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
@@ -0,0 +1,212 @@
|
||||
{
|
||||
"id": "00000000-0000-0000-0000-000000000000",
|
||||
"revision": 0,
|
||||
"last_node_id": 5,
|
||||
"last_link_id": 3,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 1,
|
||||
"type": "TransformToMatrix",
|
||||
"title": "Start pose (identity)",
|
||||
"pos": [
|
||||
-500,
|
||||
320
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "transformation matrix",
|
||||
"type": "MAT_4X4",
|
||||
"links": [
|
||||
1
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformToMatrix"
|
||||
},
|
||||
"widgets_values": [
|
||||
0.0,
|
||||
0.0,
|
||||
0.0,
|
||||
0.0,
|
||||
0.0
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "TransformToMatrix",
|
||||
"title": "End pose (edit me)",
|
||||
"pos": [
|
||||
-500,
|
||||
540
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "transformation matrix",
|
||||
"type": "MAT_4X4",
|
||||
"links": [
|
||||
2
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TransformToMatrix"
|
||||
},
|
||||
"widgets_values": [
|
||||
0.0,
|
||||
0.0,
|
||||
0.3,
|
||||
0.0,
|
||||
30.0
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"type": "CameraInterpolationNode",
|
||||
"title": "Interpolate 20 poses",
|
||||
"pos": [
|
||||
-120,
|
||||
430
|
||||
],
|
||||
"size": [
|
||||
226,
|
||||
78
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "initial_matrix",
|
||||
"type": "MAT_4X4",
|
||||
"link": 1
|
||||
},
|
||||
{
|
||||
"name": "final_matrix",
|
||||
"type": "MAT_4X4",
|
||||
"link": 2
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "trajectory",
|
||||
"type": "TENSOR",
|
||||
"links": [
|
||||
3
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "CameraInterpolationNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
20
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "SaveTrajectory",
|
||||
"title": "Save to output/*.npy",
|
||||
"pos": [
|
||||
170,
|
||||
430
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
82
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "trajectory",
|
||||
"type": "TENSOR",
|
||||
"link": 3
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "SaveTrajectory"
|
||||
},
|
||||
"widgets_values": [
|
||||
"ComfyUITrajectory"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
-1080,
|
||||
320
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
430
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Record a camera trajectory\n\nProduces the `.npy` trajectory file consumed by `PC_enricher.json` and `wan_vace_ref_to_video.json` (LoadTrajectory): two poses are SE(3)-interpolated into a smooth 20-step path and saved by **SaveTrajectory** to your ComfyUI **output** directory.\n\n- **Set:** the end pose (shift XYZ in scene units — metric if the cloud came from metric depth — plus theta = pitch, phi = yaw) and `num_steps`.\n- **Then:** move the saved file from `output/` to `input/` so LoadTrajectory can list it. A bundled example (`ComfyUITrajectory_00001.npy`) is already installed by install.py.\n- Chain more CameraInterpolationNode segments (or use CameraTrajectoryNode on a point cloud) for multi-keyframe paths."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
1,
|
||||
1,
|
||||
0,
|
||||
3,
|
||||
0,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
2,
|
||||
2,
|
||||
0,
|
||||
3,
|
||||
1,
|
||||
"MAT_4X4"
|
||||
],
|
||||
[
|
||||
3,
|
||||
3,
|
||||
0,
|
||||
4,
|
||||
0,
|
||||
"TENSOR"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
|
After Width: | Height: | Size: 46 KiB |
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"id": "4e89c8a4-594b-4d15-a089-da088966adee",
|
||||
"revision": 0,
|
||||
"last_node_id": 44,
|
||||
"last_node_id": 45,
|
||||
"last_link_id": 87,
|
||||
"nodes": [
|
||||
{
|
||||
@@ -687,7 +687,8 @@
|
||||
0,
|
||||
0,
|
||||
0
|
||||
]
|
||||
],
|
||||
"title": "Eye baseline (shiftX 0.1)"
|
||||
},
|
||||
{
|
||||
"id": 31,
|
||||
@@ -871,7 +872,8 @@
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"shifted_camera_equirect"
|
||||
]
|
||||
],
|
||||
"title": "Right eye (equirect)"
|
||||
},
|
||||
{
|
||||
"id": 27,
|
||||
@@ -908,7 +910,7 @@
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Fisheye_outpainted_flux_dev.png",
|
||||
"camera_example_fisheye.jpg",
|
||||
"image"
|
||||
]
|
||||
},
|
||||
@@ -998,7 +1000,32 @@
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"init_camera_equirect"
|
||||
]
|
||||
],
|
||||
"title": "Left eye (equirect)"
|
||||
},
|
||||
{
|
||||
"id": 45,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
929.3206787109375,
|
||||
1152.5965576171875
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
262
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# SBS VR180: synthesize the second eye\n\nFrom one 180° fisheye view, synthesizes a stereo pair: fisheye metric depth → point cloud → clean → shift the camera by the eye baseline (TransformToMatrix shiftX = 0.1) → re-project to fisheye → four chained OutpaintAnyProjection passes fill the disocclusions → both eyes are exported as 180° equirectangular images.\n\n- **Set:** fisheye input (e.g. from `Outpaint_fisheye180.json`); the baseline (0.1 ≈ 6.5 cm when depth is metric); prompts optional.\n- **Requires:** inpainting_flux (auto-installed), Depth-Anything V2.\n- **Output:** `init_camera_equirect` + `shifted_camera_equirect` — combine side-by-side for a VR180 player."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
@@ -1230,7 +1257,7 @@
|
||||
"groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"title": "Metric depth estimation",
|
||||
"title": "1. Fisheye metric depth",
|
||||
"bounding": [
|
||||
1474.8399658203125,
|
||||
1625.7159423828125,
|
||||
@@ -1243,7 +1270,7 @@
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"title": "Pointcloud manipulation",
|
||||
"title": "2. Point cloud & eye-baseline shift",
|
||||
"bounding": [
|
||||
2231.516845703125,
|
||||
1746.5635986328125,
|
||||
@@ -1256,7 +1283,7 @@
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"title": "Outpaint and project to equirect",
|
||||
"title": "3. Outpaint disocclusions & export equirect",
|
||||
"bounding": [
|
||||
3721.45654296875,
|
||||
1723.2581787109375,
|
||||
@@ -1269,7 +1296,7 @@
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"title": "Visualisations",
|
||||
"title": "Previews",
|
||||
"bounding": [
|
||||
2852.595703125,
|
||||
1071.784912109375,
|
||||
@@ -1290,7 +1317,8 @@
|
||||
-1811.182955009025
|
||||
]
|
||||
},
|
||||
"frontendVersion": "1.20.4"
|
||||
"frontendVersion": "1.20.4",
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"id": "dd56c0bf-7405-406e-924f-42b2feacb73f",
|
||||
"revision": 0,
|
||||
"last_node_id": 8,
|
||||
"last_node_id": 9,
|
||||
"last_link_id": 9,
|
||||
"nodes": [
|
||||
{
|
||||
@@ -87,7 +87,8 @@
|
||||
0,
|
||||
false,
|
||||
false
|
||||
]
|
||||
],
|
||||
"title": "Render motion"
|
||||
},
|
||||
{
|
||||
"id": 6,
|
||||
@@ -118,7 +119,8 @@
|
||||
},
|
||||
"widgets_values": [
|
||||
"ComfyUIPointCloud_00001.ply"
|
||||
]
|
||||
],
|
||||
"title": "Load saved cloud (.ply/.npy)"
|
||||
},
|
||||
{
|
||||
"id": 1,
|
||||
@@ -153,7 +155,8 @@
|
||||
0,
|
||||
0,
|
||||
0
|
||||
]
|
||||
],
|
||||
"title": "Start pose (identity)"
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
@@ -188,7 +191,8 @@
|
||||
0,
|
||||
0,
|
||||
0
|
||||
]
|
||||
],
|
||||
"title": "End pose (dolly +0.1)"
|
||||
},
|
||||
{
|
||||
"id": 8,
|
||||
@@ -228,7 +232,33 @@
|
||||
"properties": {
|
||||
"Node name for S&R": "CameraInterpolationNode"
|
||||
},
|
||||
"widgets_values": []
|
||||
"widgets_values": [
|
||||
2
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 9,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
-1097.2319946289062,
|
||||
1294.427734375
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
236
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# Load a saved point cloud & orbit\n\nReloads a point cloud saved by SavePointCloud and renders a short camera move between two poses (identity → 0.1 forward) as a WEBM.\n\n- **Set:** the file in LoadPointCloud (dropdown lists the ComfyUI input dir — run `PointCloud.json` or `fisheye_to_pointcloud.json` first); the two TransformToMatrix poses.\n- **Requires:** nothing beyond this pack."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
@@ -283,7 +313,8 @@
|
||||
-1186.3099424359816
|
||||
]
|
||||
},
|
||||
"frontendVersion": "1.21.7"
|
||||
"frontendVersion": "1.21.7",
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
}
|
||||
|
After Width: | Height: | Size: 28 KiB |
|
After Width: | Height: | Size: 41 KiB |
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"id": "0898f6a6-2814-4ccd-968a-a2405ee177e7",
|
||||
"revision": 0,
|
||||
"last_node_id": 85,
|
||||
"last_node_id": 86,
|
||||
"last_link_id": 152,
|
||||
"nodes": [
|
||||
{
|
||||
@@ -495,7 +495,8 @@
|
||||
81,
|
||||
1,
|
||||
1
|
||||
]
|
||||
],
|
||||
"title": "WAN VACE control"
|
||||
},
|
||||
{
|
||||
"id": 63,
|
||||
@@ -706,7 +707,7 @@
|
||||
"widget_ue_connectable": {}
|
||||
},
|
||||
"widgets_values": [
|
||||
"Fisheye_outpainted_flux_dev.png",
|
||||
"camera_example_fisheye.jpg",
|
||||
"image"
|
||||
]
|
||||
},
|
||||
@@ -780,7 +781,7 @@
|
||||
"widget_ue_connectable": {}
|
||||
},
|
||||
"widgets_values": [
|
||||
"wan2,1_vace14B_fp16.safetensors",
|
||||
"wan2.1_vace_14B_fp16.safetensors",
|
||||
"default"
|
||||
]
|
||||
},
|
||||
@@ -1051,7 +1052,8 @@
|
||||
0,
|
||||
true,
|
||||
false
|
||||
]
|
||||
],
|
||||
"title": "Render control frames + masks"
|
||||
},
|
||||
{
|
||||
"id": 85,
|
||||
@@ -1082,6 +1084,30 @@
|
||||
24,
|
||||
32
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 86,
|
||||
"type": "MarkdownNote",
|
||||
"title": "About this workflow",
|
||||
"pos": [
|
||||
-1085.614501953125,
|
||||
34.52389907836914
|
||||
],
|
||||
"size": [
|
||||
520,
|
||||
262
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"# WAN VACE: still image + camera move → video\n\nTurns a single 180° fisheye still into a camera-move video: metric depth → point cloud → **CameraMotionNode** renders point-splat frames and masks along a loaded trajectory; these become the VACE control video/masks with the original image as the reference, and WAN 2.1 VACE 14B synthesizes the final clip from your prompt.\n\n- **Set:** fisheye image; trajectory file (record one with SaveTrajectory); the positive prompt.\n- **Requires:** WAN 2.1 VACE models (install.sh `vae`), Depth-Anything V2, VideoHelperSuite (only for the h264/MP4 export — SaveWEBM works without it).\n- **Fixed 2026-07:** the UNET filename contained a typo (`wan2,1_vace14B` → `wan2.1_vace_14B_fp16.safetensors`)."
|
||||
],
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
@@ -1318,7 +1344,60 @@
|
||||
"IMAGE"
|
||||
]
|
||||
],
|
||||
"groups": [],
|
||||
"groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"title": "1. Fisheye image → point cloud",
|
||||
"bounding": [
|
||||
-545.614501953125,
|
||||
357.1524963378906,
|
||||
1213.47607421875,
|
||||
760.8647155761719
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"title": "2. Camera-motion control video",
|
||||
"bounding": [
|
||||
-28.916632652282715,
|
||||
970.2586669921875,
|
||||
1234.2812566757202,
|
||||
456.984619140625
|
||||
],
|
||||
"color": "#a1309b",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"title": "3. WAN VACE generation",
|
||||
"bounding": [
|
||||
-35.783010482788086,
|
||||
-25.47610092163086,
|
||||
1676.566946029663,
|
||||
968.7900657653809
|
||||
],
|
||||
"color": "#8A8",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"title": "4. Save",
|
||||
"bounding": [
|
||||
1361.3507080078125,
|
||||
107.86778259277344,
|
||||
998.1239013671875,
|
||||
914.3678741455078
|
||||
],
|
||||
"color": "#b58b2a",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
}
|
||||
],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
@@ -1334,7 +1413,8 @@
|
||||
"VHS_latentpreview": false,
|
||||
"VHS_latentpreviewrate": 0,
|
||||
"VHS_MetadataImage": true,
|
||||
"VHS_KeepIntermediate": true
|
||||
"VHS_KeepIntermediate": true,
|
||||
"camera_comfyui_rev": 1
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,633 @@
|
||||
"""World-building nodes: depth-scale anchoring, splat world enrichment along a
|
||||
trajectory (render -> outpaint -> SHARP -> align -> fuse) and panorama sphere seeding.
|
||||
|
||||
Contracts implemented here (see SPEC_4D.md):
|
||||
C4: align_depth_scale(new_depth, ref_depth, valid_mask, mode) -> (aligned, scale, shift)
|
||||
|
||||
Heavy dependencies (Flux inpainting / diffusers via OutpaintAnyProjection, SHARP)
|
||||
are only imported/loaded inside methods at call time.
|
||||
"""
|
||||
|
||||
import math
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
import torch
|
||||
from tqdm import tqdm
|
||||
|
||||
try:
|
||||
import folder_paths
|
||||
except ImportError: # Allow notebook usage outside ComfyUI
|
||||
class _FolderPathsStub:
|
||||
def __getattr__(self, name):
|
||||
raise ModuleNotFoundError(
|
||||
"folder_paths is unavailable; this node requires the ComfyUI runtime."
|
||||
)
|
||||
|
||||
folder_paths = _FolderPathsStub()
|
||||
|
||||
try:
|
||||
from . import GS_nodes as _gs
|
||||
except Exception:
|
||||
import GS_nodes as _gs
|
||||
|
||||
GaussianSplats = _gs.GaussianSplats
|
||||
Projection = _gs.Projection
|
||||
DEVICE_CHOICES = _gs.DEVICE_CHOICES
|
||||
_resolve_device_choice = _gs._resolve_device_choice
|
||||
splat_cloud_rotation = _gs.splat_cloud_rotation
|
||||
_stitch_splats = _gs._stitch_splats
|
||||
|
||||
# Zeroth-order real SH constant; rendering with add_sh_bias=True computes
|
||||
# rgb = C0 * f_dc + 0.5, so seeding uses f_dc = (rgb - 0.5) / C0.
|
||||
SH_C0 = 0.28209479177387814
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Lazy accessors for symbols provided by sibling modules / heavy dependencies
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _get_render_gaussians():
|
||||
"""Fetch GS_nodes.render_gaussians (contract C2) with an actionable error."""
|
||||
fn = getattr(_gs, "render_gaussians", None)
|
||||
if fn is None:
|
||||
raise RuntimeError(
|
||||
"GS_nodes.render_gaussians is unavailable. Update GS_nodes.py to a version "
|
||||
"that provides the module-level render_gaussians function (contract C2)."
|
||||
)
|
||||
return fn
|
||||
|
||||
|
||||
def _load_outpaint_node_class():
|
||||
"""Lazy-import OutpaintAnyProjection (pulls in Flux/diffusers machinery)."""
|
||||
try:
|
||||
from .flux_fisheye_filling_nodes import OutpaintAnyProjection
|
||||
return OutpaintAnyProjection
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
from flux_fisheye_filling_nodes import OutpaintAnyProjection
|
||||
return OutpaintAnyProjection
|
||||
except Exception as exc:
|
||||
raise RuntimeError(
|
||||
"OutpaintAnyProjection could not be imported from flux_fisheye_filling_nodes. "
|
||||
"It requires the inpainting_flux custom node package (Flux NF4 inpainting, "
|
||||
"diffusers), which this pack's install.py sets up automatically (ComfyUI-Manager "
|
||||
"runs it on install). Run install.py or fix custom_nodes/inpainting_flux. "
|
||||
f"Import error: {exc}"
|
||||
) from exc
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# C4: robust depth-scale alignment in the disparity domain
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def align_depth_scale(
|
||||
new_depth: torch.Tensor,
|
||||
ref_depth: torch.Tensor,
|
||||
valid_mask: torch.Tensor,
|
||||
mode: str = "scale_shift",
|
||||
) -> Tuple[torch.Tensor, float, float]:
|
||||
"""Least-squares scale(+shift) in DISPARITY (1/d) domain on valid_mask pixels,
|
||||
robust (clip residual outliers, 2 IRLS rounds). Returns (aligned_depth, scale, shift).
|
||||
|
||||
Fits 1/ref_depth ~= scale * (1/new_depth) + shift over valid pixels and returns
|
||||
new_depth remapped through the fitted disparity transform. If the fit is
|
||||
degenerate (too few valid pixels, non-positive/non-finite scale), returns the
|
||||
input depth unchanged with (scale=1.0, shift=0.0).
|
||||
"""
|
||||
if mode not in ("scale", "scale_shift"):
|
||||
raise ValueError(f"Unknown align mode: {mode}")
|
||||
|
||||
nd = torch.as_tensor(new_depth).float()
|
||||
# Harmonize devices: the inputs may arrive on different devices (e.g. a
|
||||
# CUDA motion mask from MotionMaskFromDepth combined with CPU depth
|
||||
# estimates); compute everything on new_depth's device.
|
||||
rd = torch.as_tensor(ref_depth).float().to(nd.device)
|
||||
vm = torch.as_tensor(valid_mask).float().to(nd.device)
|
||||
|
||||
nd_flat = nd.reshape(-1)
|
||||
rd_flat = rd.reshape(-1)
|
||||
if vm.numel() == nd_flat.numel():
|
||||
vm_flat = vm.reshape(-1)
|
||||
else:
|
||||
try:
|
||||
vm_flat = vm.expand_as(nd).reshape(-1)
|
||||
except RuntimeError as exc:
|
||||
raise ValueError(
|
||||
f"valid_mask shape {tuple(vm.shape)} is not broadcastable to depth shape {tuple(nd.shape)}"
|
||||
) from exc
|
||||
|
||||
eps = 1e-8
|
||||
valid = (
|
||||
(vm_flat > 0.5)
|
||||
& (nd_flat > eps)
|
||||
& (rd_flat > eps)
|
||||
& torch.isfinite(nd_flat)
|
||||
& torch.isfinite(rd_flat)
|
||||
)
|
||||
if int(valid.sum().item()) < 10:
|
||||
return nd.clone(), 1.0, 0.0
|
||||
|
||||
x = 1.0 / nd_flat[valid] # new disparity
|
||||
y = 1.0 / rd_flat[valid] # reference disparity
|
||||
w = torch.ones_like(x)
|
||||
|
||||
scale, shift = 1.0, 0.0
|
||||
# Initial weighted LSQ fit + 2 IRLS re-weighting rounds (outlier clipping).
|
||||
for _ in range(3):
|
||||
sw = w.sum().clamp(min=eps)
|
||||
sx = (w * x).sum()
|
||||
sy = (w * y).sum()
|
||||
if mode == "scale_shift":
|
||||
sxx = (w * x * x).sum()
|
||||
sxy = (w * x * y).sum()
|
||||
denom = sw * sxx - sx * sx
|
||||
if float(denom.abs().item()) < eps:
|
||||
s = (sxy / sxx.clamp(min=eps)).item()
|
||||
b = 0.0
|
||||
else:
|
||||
s = float(((sw * sxy - sx * sy) / denom).item())
|
||||
b = float(((sy - s * sx) / sw).item())
|
||||
else:
|
||||
sxx = (w * x * x).sum()
|
||||
sxy = (w * x * y).sum()
|
||||
s = float((sxy / sxx.clamp(min=eps)).item())
|
||||
b = 0.0
|
||||
scale, shift = s, b
|
||||
|
||||
resid = y - (scale * x + shift)
|
||||
sigma = 1.4826 * resid.abs().median()
|
||||
sigma = sigma.clamp(min=eps)
|
||||
w = (resid.abs() <= 2.5 * sigma).float()
|
||||
if float(w.sum().item()) < 10:
|
||||
break
|
||||
|
||||
if not math.isfinite(scale) or scale <= 0.0 or not math.isfinite(shift):
|
||||
return nd.clone(), 1.0, 0.0
|
||||
|
||||
disp = scale / nd.clamp(min=eps) + shift
|
||||
aligned = 1.0 / disp.clamp(min=eps)
|
||||
return aligned, float(scale), float(shift)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Internal helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _coerce_trajectory(trajectory: Any, device: torch.device) -> torch.Tensor:
|
||||
"""Coerce trajectory input to a [K,4,4] float tensor on device."""
|
||||
if isinstance(trajectory, torch.Tensor):
|
||||
traj = trajectory
|
||||
else:
|
||||
traj = torch.as_tensor(trajectory)
|
||||
traj = traj.to(device=device, dtype=torch.float32)
|
||||
if traj.dim() == 2:
|
||||
traj = traj.unsqueeze(0)
|
||||
if traj.dim() != 3 or traj.shape[-2:] != (4, 4):
|
||||
raise ValueError(f"trajectory must be [K,4,4], got shape {tuple(traj.shape)}")
|
||||
return traj
|
||||
|
||||
|
||||
def _project_to_pixels(
|
||||
xyz: torch.Tensor,
|
||||
projection: str,
|
||||
horizontal_fov: float,
|
||||
width: int,
|
||||
height: int,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
"""Project camera-frame points to integer pixel indices.
|
||||
|
||||
Returns (ix [N], iy [N], ray_depth [N], valid [N]) where valid means the point
|
||||
is in front of the camera (pinhole) and lands inside the image bounds. Uses the
|
||||
same projection math as GS_nodes rendering so pixels line up with renders.
|
||||
"""
|
||||
X, Y, Z = xyz.unbind(-1)
|
||||
if projection == "PINHOLE":
|
||||
u, v, depth = _gs._xyz_to_pinhole(X, Y, Z, horizontal_fov)
|
||||
front = Z > 1e-6
|
||||
elif projection == "FISHEYE":
|
||||
u, v, depth = _gs._xyz_to_fisheye(X, Y, Z, horizontal_fov)
|
||||
front = depth > 1e-6
|
||||
else:
|
||||
u, v, depth = _gs._xyz_to_equirect(X, Y, Z, horizontal_fov)
|
||||
front = depth > 1e-6
|
||||
|
||||
ix = torch.round((u * 0.5 + 0.5) * (width - 1)).long()
|
||||
iy = torch.round((v * 0.5 + 0.5) * (height - 1)).long()
|
||||
inside = (u >= -1.0) & (u <= 1.0) & (v >= -1.0) & (v <= 1.0)
|
||||
valid = front & inside & torch.isfinite(u) & torch.isfinite(v)
|
||||
ix = ix.clamp(0, width - 1)
|
||||
iy = iy.clamp(0, height - 1)
|
||||
return ix, iy, depth, valid
|
||||
|
||||
|
||||
def _pad_f_rest_to_order(splats: GaussianSplats, sh_order: int) -> GaussianSplats:
|
||||
"""Zero-pad SH coefficients so splats match the requested (higher) SH order.
|
||||
|
||||
Delegates to GS_nodes._pad_sh_order, which handles the renderer's
|
||||
channel-major SH layout (cat([f_dc, f_rest]).view(-1, 3, total)) correctly.
|
||||
Naively appending zeros to f_rest would shift the green/blue DC terms into
|
||||
the red channel's l>=1 slots and corrupt colors.
|
||||
"""
|
||||
return _gs._pad_sh_order(splats, sh_order)
|
||||
|
||||
|
||||
def _match_sh_orders(a: GaussianSplats, b: GaussianSplats) -> Tuple[GaussianSplats, GaussianSplats]:
|
||||
"""Bring two splat sets to a common (max) SH order via zero padding."""
|
||||
return _gs._match_sh_orders(a, b)
|
||||
|
||||
|
||||
def _scale_splats_metric(splats: GaussianSplats, factor: float) -> GaussianSplats:
|
||||
"""Uniformly rescale splat positions and sizes by a metric factor."""
|
||||
out = splats.clone()
|
||||
out.xyz = out.xyz * factor
|
||||
out.scale = out.scale + math.log(max(factor, 1e-12))
|
||||
return out
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Nodes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class DepthScaleAnchor:
|
||||
"""Aligns a depth map's scale (and optionally shift) to a reference depth map
|
||||
using a robust least-squares fit in the disparity domain (contract C4)."""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
"new_depth": ("TENSOR", {"tooltip": "Depth map to be aligned (any shape)."}),
|
||||
"ref_depth": ("TENSOR", {"tooltip": "Reference metric depth map (same shape)."}),
|
||||
"valid_mask": ("MASK", {"tooltip": "1.0 where both depths are trustworthy."}),
|
||||
"mode": (
|
||||
["scale", "scale_shift"],
|
||||
{"default": "scale_shift", "tooltip": "Fit scale only, or scale + shift, in disparity (1/d) domain."},
|
||||
),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("TENSOR", "FLOAT", "FLOAT")
|
||||
RETURN_NAMES = ("aligned_depth", "scale", "shift")
|
||||
FUNCTION = "anchor"
|
||||
CATEGORY = "Camera/World"
|
||||
DESCRIPTION = "Robustly aligns a depth map to a reference depth via disparity-domain scale(+shift)."
|
||||
|
||||
def anchor(
|
||||
self,
|
||||
new_depth: torch.Tensor,
|
||||
ref_depth: torch.Tensor,
|
||||
valid_mask: torch.Tensor,
|
||||
mode: str = "scale_shift",
|
||||
):
|
||||
aligned, scale, shift = align_depth_scale(new_depth, ref_depth, valid_mask, mode=mode)
|
||||
return (aligned, scale, shift)
|
||||
|
||||
|
||||
class SplatTrajectoryEnricher:
|
||||
"""World-expansion loop for Gaussian splats.
|
||||
|
||||
For each pose along a trajectory: render the current splats, detect uncovered
|
||||
(hole) regions, fill them with Flux outpainting, lift the filled view to new
|
||||
splats with SHARP, align the SHARP metric scale to the rendered reference
|
||||
depth, keep only the splats that cover holes, transform them to world space
|
||||
and fuse them into the running splat set.
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
choices = _gs._list_sharp_checkpoint_choices()
|
||||
return {
|
||||
"required": {
|
||||
"splats": ("GSPLAT",),
|
||||
"trajectory": ("TENSOR", {"tooltip": "[K,4,4] world-to-camera matrices of poses to visit."}),
|
||||
"camera_projection": (Projection.PROJECTIONS, {}),
|
||||
"horizontal_fov": ("FLOAT", {"default": 90.0, "min": 1.0, "max": 360.0}),
|
||||
"width": ("INT", {"default": 512, "min": 8, "max": 8192}),
|
||||
"height": ("INT", {"default": 512, "min": 8, "max": 8192}),
|
||||
"checkpoint": (
|
||||
choices,
|
||||
{
|
||||
"default": _gs._SHARP_DEFAULT_CHECKPOINT_LABEL,
|
||||
"file_chooser": True,
|
||||
"tooltip": "SHARP .pt checkpoint from the input folder, or download the default model.",
|
||||
},
|
||||
),
|
||||
"prompt": ("STRING", {"default": "", "multiline": True}),
|
||||
"num_inference_steps": ("INT", {"default": 28, "min": 10, "max": 60}),
|
||||
"guidance_scale": ("FLOAT", {"default": 5.0, "min": 0.1, "max": 30.0}),
|
||||
"mask_blur": ("INT", {"default": 5, "min": 0, "max": 512}),
|
||||
"hole_min_frac": (
|
||||
"FLOAT",
|
||||
{"default": 0.02, "min": 0.0, "max": 1.0, "step": 0.001,
|
||||
"tooltip": "Skip a view if the uncovered area is below this fraction of pixels."},
|
||||
),
|
||||
"stitch_voxel_size": ("FLOAT", {"default": 0.01, "min": 0.0, "max": 10.0}),
|
||||
"max_views": ("INT", {"default": 10, "min": 1, "max": 1000}),
|
||||
},
|
||||
"optional": {
|
||||
"device": (DEVICE_CHOICES, {"default": "auto"}),
|
||||
"cache_flux": (
|
||||
"BOOLEAN",
|
||||
{"default": True,
|
||||
"tooltip": "Keep the Flux inpainting pipeline loaded between views (avoids a multi-GB "
|
||||
"model reload per view). Disable to free VRAM after each outpaint on "
|
||||
"low-memory GPUs."},
|
||||
),
|
||||
"patch_projection": (Projection.PROJECTIONS, {"default": "PINHOLE", "tooltip": "Projection used for the outpaint patch."}),
|
||||
"patch_horiz_fov": ("FLOAT", {"default": 90.0, "min": 1.0, "max": 180.0}),
|
||||
"patch_res": ("INT", {"default": 1024, "min": 64, "max": 8192}),
|
||||
"patch_phi": ("FLOAT", {"default": 0.0, "min": -180.0, "max": 180.0}),
|
||||
"patch_theta": ("FLOAT", {"default": 0.0, "min": -90.0, "max": 90.0}),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("GSPLAT", "IMAGE", "IMAGE")
|
||||
RETURN_NAMES = ("enriched_splats", "last_render", "last_filled")
|
||||
FUNCTION = "enrich"
|
||||
CATEGORY = "Camera/World"
|
||||
DESCRIPTION = (
|
||||
"Expands a splat world along a camera trajectory: render, outpaint holes with Flux, "
|
||||
"lift with SHARP, scale-align, and smart-stitch the new content."
|
||||
)
|
||||
|
||||
@torch.no_grad()
|
||||
def enrich(
|
||||
self,
|
||||
splats: GaussianSplats,
|
||||
trajectory: torch.Tensor,
|
||||
camera_projection: str,
|
||||
horizontal_fov: float,
|
||||
width: int,
|
||||
height: int,
|
||||
checkpoint: str,
|
||||
prompt: str,
|
||||
num_inference_steps: int,
|
||||
guidance_scale: float,
|
||||
mask_blur: int,
|
||||
hole_min_frac: float,
|
||||
stitch_voxel_size: float,
|
||||
max_views: int,
|
||||
device: str = "auto",
|
||||
cache_flux: bool = True,
|
||||
patch_projection: str = "PINHOLE",
|
||||
patch_horiz_fov: float = 90.0,
|
||||
patch_res: int = 1024,
|
||||
patch_phi: float = 0.0,
|
||||
patch_theta: float = 0.0,
|
||||
) -> Tuple[GaussianSplats, torch.Tensor, torch.Tensor]:
|
||||
# Fail fast: the SHARP lift (ImageToSplat) is pinhole-only and requires
|
||||
# horizontal_fov < 179 degrees. Validating here avoids crashing in the
|
||||
# lift step AFTER minutes of rendering + Flux outpainting work.
|
||||
if not (0.0 < float(horizontal_fov) < 179.0):
|
||||
raise ValueError(
|
||||
"SplatTrajectoryEnricher lifts filled views with SHARP (pinhole), which requires "
|
||||
f"0 < horizontal_fov < 179 degrees (got {horizontal_fov}). For panoramic worlds "
|
||||
"(EQUIRECTANGULAR/FISHEYE with fov >= 179), visit several narrower pinhole poses "
|
||||
"along the trajectory instead (e.g. 90-120 degree views after SphereSplatSeed)."
|
||||
)
|
||||
render_gaussians = _get_render_gaussians()
|
||||
outpaint_cls = _load_outpaint_node_class()
|
||||
outpaint_node = outpaint_cls()
|
||||
image_to_splat = _gs.ImageToSplat()
|
||||
|
||||
target_device = _resolve_device_choice(device)
|
||||
current = splats.to(target_device) if splats.xyz.device != target_device else splats
|
||||
traj = _coerce_trajectory(trajectory, target_device)
|
||||
|
||||
if camera_projection != "PINHOLE":
|
||||
print(
|
||||
"[SplatTrajectoryEnricher] Warning: SHARP assumes pinhole geometry; "
|
||||
f"lifting filled {camera_projection} views may distort new splats."
|
||||
)
|
||||
|
||||
last_render = torch.zeros((1, height, width, 3), device=target_device)
|
||||
last_filled = torch.zeros((1, height, width, 3), device=target_device)
|
||||
added_views = 0
|
||||
|
||||
for pose in tqdm(traj[: max(1, int(max_views))], desc="Enriching splat world"):
|
||||
# 1) Render the current world from this pose.
|
||||
image, alpha, disparity = render_gaussians(
|
||||
current,
|
||||
pose,
|
||||
camera_projection,
|
||||
horizontal_fov,
|
||||
width,
|
||||
height,
|
||||
max_splats=0,
|
||||
opacity_is_logit=True,
|
||||
add_sh_bias=True,
|
||||
render_mode="auto",
|
||||
device=str(target_device).split(":")[0],
|
||||
)
|
||||
last_render = image
|
||||
|
||||
alpha_map = alpha.view(height, width).to(target_device)
|
||||
disp_map = disparity.view(height, width).to(target_device)
|
||||
hole_mask = (alpha_map < 0.5).float()
|
||||
|
||||
hole_frac = float(hole_mask.mean().item())
|
||||
if hole_frac < hole_min_frac:
|
||||
continue
|
||||
|
||||
# 2) Outpaint the uncovered region.
|
||||
filled_img, _ = outpaint_node.outpaint_any(
|
||||
image,
|
||||
input_projection=camera_projection,
|
||||
input_horiz_fov=horizontal_fov,
|
||||
output_projection=camera_projection,
|
||||
output_horiz_fov=horizontal_fov,
|
||||
output_width=width,
|
||||
output_height=height,
|
||||
patch_projection=patch_projection,
|
||||
patch_horiz_fov=patch_horiz_fov,
|
||||
patch_res=patch_res,
|
||||
patch_phi=patch_phi,
|
||||
patch_theta=patch_theta,
|
||||
prompt=prompt,
|
||||
num_inference_steps=num_inference_steps,
|
||||
# cached=True keeps the Flux NF4 pipeline resident between views
|
||||
# (cached=False forced a full multi-GB pipeline reload per view).
|
||||
cached=bool(cache_flux),
|
||||
guidance_scale=guidance_scale,
|
||||
mask_blur=mask_blur,
|
||||
mask=hole_mask.unsqueeze(0),
|
||||
debug=False,
|
||||
)
|
||||
last_filled = filled_img
|
||||
|
||||
# 3) Lift the filled view to splats in this camera frame (SHARP, metric).
|
||||
new_splats, = image_to_splat.image_to_splat(
|
||||
filled_img,
|
||||
horizontal_fov,
|
||||
checkpoint,
|
||||
device,
|
||||
)
|
||||
new_splats = new_splats.to(target_device)
|
||||
if len(new_splats) == 0:
|
||||
continue
|
||||
|
||||
# 4) Robust metric-scale alignment against the rendered reference depth.
|
||||
# Reference ray depth from the renderer: disparity = alpha / depth.
|
||||
ix, iy, sharp_depth, proj_valid = _project_to_pixels(
|
||||
new_splats.xyz, camera_projection, horizontal_fov, width, height
|
||||
)
|
||||
samp_alpha = alpha_map[iy, ix]
|
||||
samp_disp = disp_map[iy, ix]
|
||||
overlap = proj_valid & (samp_alpha >= 0.5) & (samp_disp > 1e-6) & (sharp_depth > 1e-6)
|
||||
if int(overlap.sum().item()) >= 10:
|
||||
d_ref = (samp_alpha[overlap] / samp_disp[overlap]).clamp(min=1e-6)
|
||||
ratio = d_ref / sharp_depth[overlap]
|
||||
scale_factor = float(ratio.median().item())
|
||||
if math.isfinite(scale_factor) and scale_factor > 0.0:
|
||||
new_splats = _scale_splats_metric(new_splats, scale_factor)
|
||||
|
||||
# 5) Keep only NEW content: splats whose projected pixel lies in a hole.
|
||||
samp_hole = hole_mask[iy, ix]
|
||||
keep = proj_valid & (samp_hole > 0.5)
|
||||
if not bool(keep.any().item()):
|
||||
continue
|
||||
new_splats = new_splats[keep]
|
||||
|
||||
# 6) Camera frame -> world frame (pose is world-to-camera).
|
||||
new_world = splat_cloud_rotation(new_splats, torch.inverse(pose))
|
||||
|
||||
# 7) Fuse into the running world. Concatenation is cheap; the full
|
||||
# smart voxel reduce is deferred to a single pass after the loop,
|
||||
# so each view does not re-copy and re-unique-sort the entire
|
||||
# accumulated cloud (O(views x N) work/memory otherwise).
|
||||
cur_m, new_m = _match_sh_orders(current, new_world)
|
||||
current = _gs._concat_splats([cur_m, new_m])
|
||||
added_views += 1
|
||||
|
||||
if added_views > 0 and stitch_voxel_size > 0.0:
|
||||
current = _stitch_splats([current], "smart", stitch_voxel_size, 5.0)
|
||||
|
||||
return (current, last_render, last_filled)
|
||||
|
||||
|
||||
class SphereSplatSeed:
|
||||
"""Seeds a 360-degree splat world from an equirectangular panorama: one Gaussian
|
||||
per (subsampled) pixel, placed on a depth sphere around the origin."""
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls) -> Dict[str, Any]:
|
||||
return {
|
||||
"required": {
|
||||
"image": ("IMAGE", {"tooltip": "Equirectangular panorama [1,H,W,3]."}),
|
||||
"horizontal_fov": ("FLOAT", {"default": 360.0, "min": 1.0, "max": 360.0}),
|
||||
"radius": ("FLOAT", {"default": 5.0, "min": 0.01, "max": 10000.0, "tooltip": "Sphere radius used when no depth map is provided."}),
|
||||
"splat_scale_frac": (
|
||||
"FLOAT",
|
||||
{"default": 1.5, "min": 0.1, "max": 10.0,
|
||||
"tooltip": "Splat sigma as a fraction of the local point spacing (larger = smoother, fewer holes)."},
|
||||
),
|
||||
"stride": ("INT", {"default": 2, "min": 1, "max": 64, "tooltip": "Pixel subsampling stride (1 Gaussian per stride x stride block)."}),
|
||||
},
|
||||
"optional": {
|
||||
"depth": ("TENSOR", {"tooltip": "Optional ray-depth map [H,W] (or [1,H,W]/[H,W,1]) matching the panorama."}),
|
||||
"opacity_logit": ("FLOAT", {"default": 6.0, "min": -10.0, "max": 20.0}),
|
||||
"device": (DEVICE_CHOICES, {"default": "auto"}),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("GSPLAT",)
|
||||
RETURN_NAMES = ("splats",)
|
||||
FUNCTION = "seed_sphere"
|
||||
CATEGORY = "Camera/World"
|
||||
DESCRIPTION = "Converts an equirectangular panorama into a Gaussian sphere seeding a 360-degree world."
|
||||
|
||||
@torch.no_grad()
|
||||
def seed_sphere(
|
||||
self,
|
||||
image: torch.Tensor,
|
||||
horizontal_fov: float = 360.0,
|
||||
radius: float = 5.0,
|
||||
splat_scale_frac: float = 1.5,
|
||||
stride: int = 2,
|
||||
depth: Optional[torch.Tensor] = None,
|
||||
opacity_logit: float = 6.0,
|
||||
device: str = "auto",
|
||||
) -> Tuple[GaussianSplats]:
|
||||
target_device = _resolve_device_choice(device)
|
||||
|
||||
img = image
|
||||
if img.dim() == 4:
|
||||
img = img[0]
|
||||
if img.dim() != 3 or img.shape[-1] < 3:
|
||||
raise ValueError(f"Expected IMAGE [1,H,W,3], got shape {tuple(image.shape)}")
|
||||
img = img[..., :3].to(device=target_device, dtype=torch.float32)
|
||||
H, W = int(img.shape[0]), int(img.shape[1])
|
||||
|
||||
depth_map = None
|
||||
if depth is not None:
|
||||
d = torch.as_tensor(depth).to(device=target_device, dtype=torch.float32)
|
||||
if d.dim() == 3:
|
||||
# [1,H,W], [T,H,W] (take first) or [H,W,1]
|
||||
d = d[..., 0] if d.shape[-1] == 1 else d[0]
|
||||
if d.dim() != 2:
|
||||
raise ValueError(f"depth must reduce to [H,W], got shape {tuple(depth.shape)}")
|
||||
if d.shape != (H, W):
|
||||
d = torch.nn.functional.interpolate(
|
||||
d.unsqueeze(0).unsqueeze(0), size=(H, W), mode="bilinear", align_corners=True
|
||||
)[0, 0]
|
||||
depth_map = d.clamp(min=1e-6)
|
||||
|
||||
stride = max(1, int(stride))
|
||||
ys = torch.arange(0, H, stride, device=target_device)
|
||||
xs = torch.arange(0, W, stride, device=target_device)
|
||||
yy, xx = torch.meshgrid(ys, xs, indexing="ij")
|
||||
yy = yy.reshape(-1)
|
||||
xx = xx.reshape(-1)
|
||||
|
||||
# Match the renderer's equirect mapping (GS_nodes._xyz_to_equirect):
|
||||
# u = lon / (fov_rad/2), v = lat / (pi/2), px = (u*0.5+0.5)*(W-1)
|
||||
fov_rad = math.radians(horizontal_fov)
|
||||
u = xx.float() / max(W - 1, 1) * 2.0 - 1.0
|
||||
v = yy.float() / max(H - 1, 1) * 2.0 - 1.0
|
||||
lon = u * (fov_rad / 2.0)
|
||||
lat = v * (math.pi / 2.0)
|
||||
|
||||
if depth_map is not None:
|
||||
d = depth_map[yy, xx]
|
||||
else:
|
||||
d = torch.full_like(lon, float(radius))
|
||||
|
||||
cos_lat = torch.cos(lat)
|
||||
X = d * cos_lat * torch.sin(lon)
|
||||
Y = d * torch.sin(lat)
|
||||
Z = d * cos_lat * torch.cos(lon)
|
||||
xyz = torch.stack([X, Y, Z], dim=-1)
|
||||
|
||||
rgb = img[yy, xx, :]
|
||||
# Rendering with add_sh_bias=True evaluates rgb = C0 * f_dc + 0.5.
|
||||
f_dc = (rgb - 0.5) / SH_C0
|
||||
|
||||
# Isotropic sigma from local angular spacing (radians per sample) times depth.
|
||||
ang_spacing = float(stride) * max(fov_rad / max(W, 1), math.pi / max(H, 1))
|
||||
sigma = (splat_scale_frac * ang_spacing * d).clamp(min=1e-6)
|
||||
scale = torch.log(sigma).unsqueeze(-1).expand(-1, 3).contiguous()
|
||||
|
||||
n = xyz.shape[0]
|
||||
rotation = torch.zeros((n, 4), device=target_device, dtype=torch.float32)
|
||||
rotation[:, 0] = 1.0 # identity wxyz quaternion
|
||||
opacity = torch.full((n, 1), float(opacity_logit), device=target_device, dtype=torch.float32)
|
||||
f_rest = torch.zeros((n, 0), device=target_device, dtype=torch.float32)
|
||||
|
||||
splats = GaussianSplats(
|
||||
xyz=xyz,
|
||||
scale=scale,
|
||||
rotation=rotation,
|
||||
opacity=opacity,
|
||||
f_dc=f_dc,
|
||||
f_rest=f_rest,
|
||||
sh_order=0,
|
||||
)
|
||||
return (splats,)
|
||||
|
||||
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"DepthScaleAnchor": DepthScaleAnchor,
|
||||
"SplatTrajectoryEnricher": SplatTrajectoryEnricher,
|
||||
"SphereSplatSeed": SphereSplatSeed,
|
||||
}
|
||||