diff --git a/README.md b/README.md index 6222f68..59ee74d 100644 --- a/README.md +++ b/README.md @@ -11,6 +11,15 @@ It can be directly connected to the WanVideo ATI Tracks Node. ### 2025-6-4 1st commit +### 2025-6-6 +added utility node +- PerlinCoordinateRandomizerNode +Applies Perlin noise-based randomization to coordinate data, adding natural, smooth variations to tracking points across frames. +- XYMotionAmplifierNode +Amplifies coordinate movement with directional control for X/Y axes, preserving static points while enhancing motion intensity with optional mask-based selection. +- GridPointGeneratorNode +Generates a grid of coordinate points. + ### Related resources - [CoTracker](https://github.com/facebookresearch/co-tracker) - [ComfyUI-WanVideoWrapper](https://github.com/kijai/ComfyUI-WanVideoWrapper) diff --git a/__init__.py b/__init__.py index ab9b3fb..7c478f2 100644 --- a/__init__.py +++ b/__init__.py @@ -1,2 +1,13 @@ from .cotracker_node import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS +from .perlin_noise_node import NODE_CLASS_MAPPINGS as PERLIN_NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as PERLIN_NODE_DISPLAY_NAME_MAPPINGS +from .utility_node import NODE_CLASS_MAPPINGS as UTILITY_NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as UTILITY_NODE_DISPLAY_NAME_MAPPINGS + + +NODE_CLASS_MAPPINGS.update(PERLIN_NODE_CLASS_MAPPINGS) +NODE_CLASS_MAPPINGS.update(UTILITY_NODE_CLASS_MAPPINGS) + +NODE_DISPLAY_NAME_MAPPINGS.update(PERLIN_NODE_DISPLAY_NAME_MAPPINGS) +NODE_DISPLAY_NAME_MAPPINGS.update(UTILITY_NODE_DISPLAY_NAME_MAPPINGS) + + __all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"] diff --git a/cotracker_node.py b/cotracker_node.py index ca72b9f..5e8c793 100644 --- a/cotracker_node.py +++ b/cotracker_node.py @@ -31,7 +31,8 @@ class CoTrackerNode: "default": 20, "min": 0, "max": 100, - "step": 1 + "step": 1, + "tooltip": "Number of divisions along both width and height to create a grid of tracking points." }), "max_num_of_points": ("INT", { "default": 100, @@ -52,7 +53,8 @@ class CoTrackerNode: "default": 30, "min": 0, "max": 500, - "step": 1 + "step": 1, + "tooltip": "Minimum distance between tracking points" }), "force_offload": ("BOOLEAN", {"default": True}), } @@ -138,6 +140,11 @@ class CoTrackerNode: queries = self.prepare_query_points(points, video.shape) + if video.shape[1] <= self.model.step: + print(f"{video.shape[1]=}") + raise ValueError(f"At least {self.model.step+1} frames are required to perform tracking.") + + results = [] if len(points) > 0: diff --git a/perlin_noise_node.py b/perlin_noise_node.py new file mode 100644 index 0000000..f51ab0d --- /dev/null +++ b/perlin_noise_node.py @@ -0,0 +1,286 @@ +import numpy as np +import math +import json +import cv2 +import torch + +class PerlinNoise: + """ + Simple Perlin noise implementation for coordinate randomization + """ + def __init__(self, seed=None): + if seed is not None: + np.random.seed(seed) + + # Generate permutation table + self.p = np.arange(256) + np.random.shuffle(self.p) + self.p = np.concatenate([self.p, self.p]) # Duplicate for overflow handling + + def fade(self, t): + """Fade function for smooth interpolation""" + return t * t * t * (t * (t * 6 - 15) + 10) + + def lerp(self, t, a, b): + """Linear interpolation""" + return a + t * (b - a) + + def grad(self, hash_val, x, y, z): + """Gradient function""" + h = hash_val & 15 + u = x if h < 8 else y + v = y if h < 4 else (x if h == 12 or h == 14 else z) + return (u if (h & 1) == 0 else -u) + (v if (h & 2) == 0 else -v) + + def noise(self, x, y, z): + """Generate 3D Perlin noise""" + # Find unit cube containing point + X = int(math.floor(x)) & 255 + Y = int(math.floor(y)) & 255 + Z = int(math.floor(z)) & 255 + + # Find relative position in cube + x -= math.floor(x) + y -= math.floor(y) + z -= math.floor(z) + + # Compute fade curves + u = self.fade(x) + v = self.fade(y) + w = self.fade(z) + + # Hash coordinates of cube corners + A = self.p[X] + Y + AA = self.p[A] + Z + AB = self.p[A + 1] + Z + B = self.p[X + 1] + Y + BA = self.p[B] + Z + BB = self.p[B + 1] + Z + + # Interpolate between cube corners + return self.lerp(w, + self.lerp(v, + self.lerp(u, self.grad(self.p[AA], x, y, z), + self.grad(self.p[BA], x-1, y, z)), + self.lerp(u, self.grad(self.p[AB], x, y-1, z), + self.grad(self.p[BB], x-1, y-1, z))), + self.lerp(v, + self.lerp(u, self.grad(self.p[AA+1], x, y, z-1), + self.grad(self.p[BA+1], x-1, y, z-1)), + self.lerp(u, self.grad(self.p[AB+1], x, y-1, z-1), + self.grad(self.p[BB+1], x-1, y-1, z-1)))) + +def randomize_coordinates_with_perlin(coord_data, + spatial_scale=10.0, + time_scale=50.0, + intensity=1.0, + octaves=3, + seed=None, + mask=None): + """ + Randomize coordinate data using 3D Perlin noise + + Parameters: + coord_data: list of lists - [[(x1,y1), (x2,y2), ...], [(x1,y1), (x2,y2), ...], ...] + Each inner list contains all frames for one coordinate point + spatial_scale: float - spatial frequency of noise (larger = smoother in space) + time_scale: float - temporal frequency of noise (larger = slower changes) + intensity: float - amplitude of noise displacement + octaves: int - number of noise octaves to combine (more = more detail) + seed: int - random seed for reproducibility + + Returns: + randomized_data: randomized coordinate data in the same format (with int coordinates) + """ + + # Initialize Perlin noise generator + perlin = PerlinNoise(seed=seed) + + # Get data dimensions + num_points = len(coord_data) + num_frames = len(coord_data[0]) + + print(f"Data shape: {num_points} coordinate points, {num_frames} frames each") + print(f"Parameters: spatial_scale={spatial_scale}, time_scale={time_scale}, intensity={intensity}, octaves={octaves}") + + # Convert to numpy array for easier processing [point, frame, xy] + coords_array = np.array(coord_data, dtype=float) + + def multi_octave_noise(x, y, z, octaves): + """Generate multi-octave Perlin noise""" + value = 0 + amplitude = 1 + frequency = 1 + max_value = 0 + + for _ in range(octaves): + value += perlin.noise(x * frequency, y * frequency, z * frequency) * amplitude + max_value += amplitude + amplitude *= 0.5 + frequency *= 2 + + return value / max_value + + def is_masked(x, y): + if mask is None: + return True # no mask + return (0 <= int(x) < mask.shape[1] and + 0 <= int(y) < mask.shape[0] and + mask[int(y), int(x)] > 0) + + # Generate noise for each coordinate point and frame + randomized_coords = coords_array.copy() + + for point_idx in range(num_points): + initial_x, initial_y = coords_array[point_idx, 0] + + if is_masked(initial_x, initial_y): + for frame_idx in range(num_frames): + # Current position + curr_x, curr_y = coords_array[point_idx, frame_idx] + + # Time coordinate + t = frame_idx / time_scale + + # Generate noise using current position for spatial coherence + noise_x = multi_octave_noise(curr_x / spatial_scale, + curr_y / spatial_scale, + t, octaves) * intensity + + # Offset y-noise sampling to decorrelate from x-noise + noise_y = multi_octave_noise((curr_x + 1000) / spatial_scale, + curr_y / spatial_scale, + t, octaves) * intensity + + # Apply noise + new_x = curr_x + noise_x + new_y = curr_y + noise_y + + # Convert back to integers + randomized_coords[point_idx, frame_idx, 0] = round(new_x) + randomized_coords[point_idx, frame_idx, 1] = round(new_y) + + + # Convert back to original format with integer coordinates + randomized_data = [ + [(int(randomized_coords[point, frame, 0]), int(randomized_coords[point, frame, 1])) + for frame in range(num_frames)] + for point in range(num_points) + ] + + return randomized_data + + + + +class PerlinCoordinateRandomizerNode: + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "tracking_results": ("STRING",), + }, + "optional": { + "images_for_marker": ("IMAGE", {"default": None}), + "noise_mask": ("MASK", {"tooltip": "Mask for randomize"}), + "spatial_scale": ("INT", { + "default": 1000, + "min": 1, + "max": 9999, + "step": 1, + "tooltip": "spatial_scale (pixels) / Larger → Smooth, coherent movement (nearby points move similarly) / Smaller → Chaotic, erratic movement (neighboring points move randomly)" + }), + "time_scale": ("INT", { + "default": 60, + "min": 1, + "max": 1000, + "step": 1, + "tooltip": "time_scale (frames) / Larger → Slow movement / Smaller → Fast movement" + }), + "intensity": ("INT", { + "default": 100, + "min": 1, + "max": 1000, + "step": 1, + "tooltip": "intensity (pixels) / Larger → Big displacement / Smaller → Small displacement" + }), + "octaves": ("INT", { + "default": 3, + "min": 1, + "max": 10, + "step": 1, + "tooltip": "octaves (layers) / Larger → Complex, detailed movement / Smaller → Simple, basic movement" + }), + "seed": ("INT", {"default": 0, "min": 0, "max": 0xffffffff}), + "enabled": ("BOOLEAN", {"default": True}), + } + } + + RETURN_TYPES = ("STRING","IMAGE") + RETURN_NAMES = ("randomized_results","image_with_results") + FUNCTION = "apply_perlin_noise" + CATEGORY = "tracking/utility" + + def apply_perlin_noise(self, tracking_results, images_for_marker=None, noise_mask=None, spatial_scale=1000, time_scale=60, intensity=100, octaves=3, seed=42, enabled=True): + + if enabled == False: + return (tracking_results, images_for_marker) + + if noise_mask is not None: + noise_mask = noise_mask.cpu().numpy() + if len(noise_mask.shape) == 3 and noise_mask.shape[0] == 1: + noise_mask = noise_mask[0] + + + raw_data = [[(d["x"], d["y"]) for d in json.loads(s)[0]] for s in tracking_results] + + # Apply Perlin noise randomization + randomized_data = randomize_coordinates_with_perlin( + raw_data, + spatial_scale=spatial_scale, # spatial smoothness (larger = smoother) + time_scale=time_scale, # temporal smoothness (larger = slower changes) + intensity=intensity, # noise amplitude + octaves=octaves, # noise detail levels + seed=seed, # for reproducibility + mask=noise_mask + ) + + if images_for_marker is not None: + images_with_markers = self.apply_marker(randomized_data, images_for_marker) + else: + images_with_markers = None + + result = [json.dumps([[{"x": x, "y": y} for x, y in coords]]) for coords in randomized_data] + + return (result, images_with_markers) + + def apply_marker(self, randomized_data, images): + + images_np = images.cpu().numpy() + images_np = (images_np * 255).astype(np.uint8) + + marker_radius = 3 + marker_thickness = -1 + marker_color = (0, 0, 255) + + for coords in randomized_data: + for i,(x,y) in enumerate(coords): + if i < images_np.shape[0]: + cv2.circle(images_np[i], (int(x), int(y)), marker_radius, marker_color, marker_thickness) + + images_with_markers = torch.from_numpy(images_np) + images_with_markers = images_with_markers.float() / 255.0 + + return images_with_markers + + + +NODE_CLASS_MAPPINGS = { + "PerlinCoordinateRandomizerNode": PerlinCoordinateRandomizerNode +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "PerlinCoordinateRandomizerNode": "PerlinNoise Coordinate Randomizer" +} + diff --git a/utility_node.py b/utility_node.py new file mode 100644 index 0000000..a8d6210 --- /dev/null +++ b/utility_node.py @@ -0,0 +1,226 @@ +import numpy as np +import json +import torch +import cv2 + + +class GridPointGeneratorNode: + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "image": ("IMAGE", {"default": None}), + "grid_size": ("INT", { + "default": 10, + "min": 1, + "max": 1000, + "step": 1, + "tooltip": "Number of divisions along both width and height to create a grid of tracking points." + }), + "frame_count": ("INT", { + "default": 121, + "min": 1, + "max": 9999, + "step": 1, + }), + }, + "optional": { + "mask": ("MASK", {"tooltip": "Generate grid points only inside masked area"}), + "existing_coordinates": ("STRING",), + } + } + + RETURN_TYPES = ("STRING",) + RETURN_NAMES = ("grid_coordinates","") + FUNCTION = "generate_grid" + CATEGORY = "tracking/utility" + + def generate_grid(self, image, grid_size=10, frame_count=121, mask=None, existing_coordinates=""): + + # (B, H, W, C) + _, H, W, _ = image.shape + + if mask is not None: + mask = mask.cpu().numpy() + if len(mask.shape) == 3 and mask.shape[0] == 1: + mask = mask[0] + + raw_data = [] + if existing_coordinates and len(existing_coordinates) > 0: + raw_data = [[(d["x"], d["y"]) for d in json.loads(s)[0]] for s in existing_coordinates] + + # Generate grid points + grid_points = [] + step_x = W / (grid_size + 1) # +1 to avoid edge placement + step_y = H / (grid_size + 1) + + for i in range(1, grid_size + 1): + for j in range(1, grid_size + 1): + x = int(i * step_x) + y = int(j * step_y) + + # Check if point is within mask (if mask is provided) + if mask is not None: + if y < mask.shape[0] and x < mask.shape[1]: + if mask[y, x] > 0: + grid_points.append((x, y)) + else: + continue + else: + grid_points.append((x, y)) + + # Add grid points to raw_data (each grid point gets all frames) + for grid_point in grid_points: + point_frames = [grid_point for _ in range(frame_count)] + raw_data.append(point_frames) + + + result = [json.dumps([[{"x": x, "y": y} for x, y in coords]]) for coords in raw_data] + + return (result,) + + +class XYMotionAmplifierNode: + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "coordinates": ("STRING",), + "x_positive_amp": ("FLOAT", { + "default": 1.0, + "min": 0.0, + "max": 100.0, + "step": 0.1, + }), + "x_negative_amp": ("FLOAT", { + "default": 1.0, + "min": 0.0, + "max": 100.0, + "step": 0.1, + }), + "y_positive_amp": ("FLOAT", { + "default": 1.0, + "min": 0.0, + "max": 100.0, + "step": 0.1, + }), + "y_negative_amp": ("FLOAT", { + "default": 1.0, + "min": 0.0, + "max": 100.0, + "step": 0.1, + }), + }, + "optional": { + "mask": ("MASK", {"tooltip": "Modify points only inside masked area"}), + "images_for_marker": ("IMAGE", {"default": None}), + } + } + + RETURN_TYPES = ("STRING","IMAGE") + RETURN_NAMES = ("coordinates","image_with_results") + FUNCTION = "amplify" + CATEGORY = "tracking/utility" + + + def amplify(self, coordinates, x_positive_amp, x_negative_amp, y_positive_amp, y_negative_amp, mask=None, images_for_marker=None): + + if mask is not None: + mask = mask.cpu().numpy() + if len(mask.shape) == 3 and mask.shape[0] == 1: + mask = mask[0] + + raw_data = [[(d["x"], d["y"]) for d in json.loads(s)[0]] for s in coordinates] + + amplified_data = [] + for point_idx, point_frames in enumerate(raw_data): + + should_amplify = True + if mask is not None and len(point_frames) > 0: + initial_x, initial_y = point_frames[0] + # Convert to integer coordinates for mask indexing + mask_x = int(round(initial_x)) + mask_y = int(round(initial_y)) + # Check bounds and mask value + if (0 <= mask_y < mask.shape[0] and 0 <= mask_x < mask.shape[1]): + should_amplify = mask[mask_y, mask_x] > 0 + else: + should_amplify = False + + amplified_point_frames = [] + for frame_idx, (x, y) in enumerate(point_frames): + if frame_idx == 0 or not should_amplify: + # First frame or point not in mask: no amplification + new_x, new_y = x, y + else: + # Calculate movement from previous frame + prev_x, prev_y = point_frames[frame_idx - 1] + delta_x = x - prev_x + delta_y = y - prev_y + + # Amplify movement and add to previous amplified position + if delta_x > 0: + amplified_delta_x = delta_x * x_positive_amp + elif delta_x < 0: + amplified_delta_x = delta_x * x_negative_amp + else: + amplified_delta_x = 0 + + if delta_y > 0: + amplified_delta_y = delta_y * y_positive_amp + elif delta_y < 0: + amplified_delta_y = delta_y * y_negative_amp + else: + amplified_delta_y = 0 + + prev_amplified_x, prev_amplified_y = amplified_point_frames[frame_idx - 1] + new_x = prev_amplified_x + amplified_delta_x + new_y = prev_amplified_y + amplified_delta_y + + amplified_point_frames.append((new_x, new_y)) + + amplified_data.append(amplified_point_frames) + + + if images_for_marker is not None: + images_with_markers = self.apply_marker(amplified_data, images_for_marker) + else: + images_with_markers = None + + + result = [json.dumps([[{"x": x, "y": y} for x, y in coords]]) for coords in amplified_data] + + return (result,images_with_markers) + + def apply_marker(self, amplified_data, images): + + images_np = images.cpu().numpy() + images_np = (images_np * 255).astype(np.uint8) + + marker_radius = 3 + marker_thickness = -1 + marker_color = (0, 0, 255) + + for coords in amplified_data: + for i,(x,y) in enumerate(coords): + if i < images_np.shape[0]: + cv2.circle(images_np[i], (int(x), int(y)), marker_radius, marker_color, marker_thickness) + + images_with_markers = torch.from_numpy(images_np) + images_with_markers = images_with_markers.float() / 255.0 + + return images_with_markers + + +NODE_CLASS_MAPPINGS = { + "GridPointGeneratorNode": GridPointGeneratorNode, + "XYMotionAmplifierNode": XYMotionAmplifierNode +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "GridPointGeneratorNode": "Grid Point Generator", + "XYMotionAmplifierNode": "XY Motion Amplifier" +} +