commit 358b4b5889a5025ab87490f674403e99ffb88eaa Author: Nanthakumar Date: Sat Jul 26 09:18:11 2025 +0530 first commit diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..de35933 --- /dev/null +++ b/.gitignore @@ -0,0 +1,2 @@ +__pycache__/ +**/__pycache__/ diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..715bf54 --- /dev/null +++ b/LICENSE @@ -0,0 +1,14 @@ +Creative Commons Attribution-NonCommercial 4.0 International License + +This work is licensed under the Creative Commons Attribution-NonCommercial 4.0 International License. +To view a copy of this license, visit http://creativecommons.org/licenses/by-nc/4.0/ + +You are free to: +- Share — copy and redistribute the material in any medium or format +- Adapt — remix, transform, and build upon the material + +Under the following terms: +- Attribution — You must give appropriate credit, provide a link to the license, and indicate if changes were made. +- NonCommercial — You may not use the material for commercial purposes. + +No warranties are given. The license may not give you all of the permissions necessary for your intended use. diff --git a/NOTICE.txt b/NOTICE.txt new file mode 100644 index 0000000..48594b2 --- /dev/null +++ b/NOTICE.txt @@ -0,0 +1,10 @@ +📛 Non-Commercial License Notice + +This codebase is licensed under the Creative Commons Attribution-NonCommercial 4.0 International License (CC BY-NC 4.0). + +🔒 You MAY NOT: +- Use this in any commercial project or SaaS service +- Include it in paid marketplaces or plugins +- Monetize this codebase directly or indirectly without the author's explicit consent + +For commercial use, please contact the author at GitHub: https://github.com/YOUR_USERNAME diff --git a/README.md b/README.md new file mode 100644 index 0000000..be2a16d --- /dev/null +++ b/README.md @@ -0,0 +1,69 @@ +# 🎛️ Rndnanthu ComfyUI Custom Nodes + +A curated collection of high-performance, creator-centric **ComfyUI nodes** built for image nerds, colorists, VFX artists, and generative AI creators. +🧠 Powered by love, LUTs, grain, and deep latent dreams. 🌀 + +> ⚠️ **NOTE:** Licensed strictly for **non-commercial use** under [CC BY-NC 4.0](./LICENSE) + + +## 🔧 Node Categories + +### 🎨 `colortools/` – Color Tools for Artists +- **`log_color_conversion`** – Convert between LOG/Linear/Rec.709 using LUTs (.cube) +- **`ColorGradingNode`** – Lift-Gamma-Gain, Offset/Slope color grading +- **`colorspacesim`** – Simulate camera profiles (S-Log, V-Log, Cine etc.) + LUT auto-detect +- **`autogradepro`** – Auto exposure, contrast, ISO, temperature correction +- **`ColorAnalysisPlotNode`** – RGB Parade, Histogram, Vectorscope, False Color *(experimental)* + + +### 🎞️ `grain/` – Film & Texture Simulation +- **`FilmGrain`** – Organic, preset-based grain generation (with random seed) + + +### ✨ `prompt/` – Prompt Logic +- **`PromptGenerator`** – Smart multimodal prompt builder for images & videos + + +## 📦 Installation + +```bash +git clone https://github.com/rndnanthu/Comfyui-rndnanthu.git + +pip install -r requirements.txt +```` + +Then restart **ComfyUI**. + + +## 🛑 License: [CC BY-NC 4.0](https://creativecommons.org/licenses/by-nc/4.0/) + +This repository is released under the **Creative Commons Attribution-NonCommercial 4.0 International License**. + +✅ **Allowed**: + +* Personal use +* Educational and research use +* Modifications for private workflows + +🚫 **NOT Allowed**: + +* Resale or bundling in paid plugin packs +* Use in SaaS/commercial tools without permission +* Monetized redistribution + +> 💬 Want to use it commercially? [Open an issue](https://github.com/rndnanthu/Comfyui-rndnanthu/issues) or contact me via GitHub. + + +## 👑 Author +Made with ❤️ by **RNDNANTHU** + + +## 📌 Connect With Me +- 🔗 GitHub: [@rndnanthu](https://github.com/rndnanthu) +- 🎥 YouTube Channels: + - [@rndnanthu](https://www.youtube.com/@rndwithnanthu) + - [@CGKalvi](https://www.youtube.com/@CGKalvi) + + +> Built for artists who love precision, grain, and control. + diff --git a/__init__.py b/__init__.py new file mode 100644 index 0000000..0f48046 --- /dev/null +++ b/__init__.py @@ -0,0 +1,42 @@ +# © 2025 rndnanthu – Licensed under CC BY-NC 4.0 (https://creativecommons.org/licenses/by-nc/4.0/) +# --- Core Node Imports --- +from .prompt.PromptGenerator import NODE_CLASS_MAPPINGS as PROMPT_NODES, NODE_DISPLAY_NAME_MAPPINGS as PROMPT_DISPLAY +from .colortools.log_color_conversion import NODE_CLASS_MAPPINGS as LOG_NODES, NODE_DISPLAY_NAME_MAPPINGS as LOG_DISPLAY +from .colortools.ColorGradingNode import NODE_CLASS_MAPPINGS as COLOR_NODES, NODE_DISPLAY_NAME_MAPPINGS as COLOR_DISPLAY +from .grain.FilmGrain import NODE_CLASS_MAPPINGS as GRAIN_NODES, NODE_DISPLAY_NAME_MAPPINGS as GRAIN_DISPLAY + +# --- Custom Visual Nodes --- +from .colortools.colorspacesim import NODE_CLASS_MAPPINGS as COLORSPACE_NODES, NODE_DISPLAY_NAME_MAPPINGS as COLORSPACE_DISPLAY +from .colortools.autogradepro import NODE_CLASS_MAPPINGS as AUTOGRADE_NODES, NODE_DISPLAY_NAME_MAPPINGS as AUTOGRADE_DISPLAY +from .colortools.ColorAnalysisPlotNode import NODE_CLASS_MAPPINGS as COLORANALYSIS_NODES, NODE_DISPLAY_NAME_MAPPINGS as COLORPLOT_DISPLAY + + +# === Combine All Node Mappings === +NODE_CLASS_MAPPINGS = {} +NODE_CLASS_MAPPINGS.update(PROMPT_NODES) +NODE_CLASS_MAPPINGS.update(LOG_NODES) +NODE_CLASS_MAPPINGS.update(COLOR_NODES) +NODE_CLASS_MAPPINGS.update(GRAIN_NODES) +NODE_CLASS_MAPPINGS.update(COLORSPACE_NODES) +NODE_CLASS_MAPPINGS.update(AUTOGRADE_NODES) +NODE_CLASS_MAPPINGS.update(COLORANALYSIS_NODES) + +NODE_DISPLAY_NAME_MAPPINGS = {} +NODE_DISPLAY_NAME_MAPPINGS.update(PROMPT_DISPLAY) +NODE_DISPLAY_NAME_MAPPINGS.update(LOG_DISPLAY) +NODE_DISPLAY_NAME_MAPPINGS.update(COLOR_DISPLAY) +NODE_DISPLAY_NAME_MAPPINGS.update(GRAIN_DISPLAY) +NODE_DISPLAY_NAME_MAPPINGS.update(COLORSPACE_DISPLAY) +NODE_DISPLAY_NAME_MAPPINGS.update(AUTOGRADE_DISPLAY) +NODE_DISPLAY_NAME_MAPPINGS.update(COLORPLOT_DISPLAY) + +# === Final Exports === +__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS'] + + +YELLOW = '\033[93m' +WHITE ='\033[37m' +RESET = '\033[0m' + +print(f'{YELLOW}[RNDNANTHU] : {WHITE}THANKS FOR USING{RESET}') +print(f'{YELLOW}[RNDNANTHU] : {WHITE}KINDLY GIVE YOUR SUPPORT TO {YELLOW}RNDNANTHU{RESET}') diff --git a/assets/luts/put_luts_here b/assets/luts/put_luts_here new file mode 100644 index 0000000..e69de29 diff --git a/assets/presets/film_grain_presets.json b/assets/presets/film_grain_presets.json new file mode 100644 index 0000000..6f53d87 --- /dev/null +++ b/assets/presets/film_grain_presets.json @@ -0,0 +1,14 @@ +{ + "Kodak_Vision3_250D": { "strength": 0.18, "grain_size": 1.2, "color_var": 0.08, "sharpen": 0.15 }, + "Kodak_TriX_400": { "strength": 0.25, "grain_size": 2.2, "color_var": 0.0, "sharpen": 0.05 }, + "Fujifilm_Superia_400": { "strength": 0.15, "grain_size": 1.0, "color_var": 0.1, "sharpen": 0.0 }, + "Fujifilm_Provia_100": { "strength": 0.12, "grain_size": 0.8, "color_var": 0.05, "sharpen": 0.0 }, + "Ilford_HP5_Plus_400": { "strength": 0.3, "grain_size": 2.5, "color_var": 0.0, "sharpen": 0.1 }, + "Kodak_Portra_400": { "strength": 0.14, "grain_size": 1.0, "color_var": 0.12, "sharpen": 0.05 }, + "Kodak_Tmax_3200": { "strength": 0.35, "grain_size": 3.0, "color_var": 0.0, "sharpen": 0.2 }, + "Fujifilm_Astia_100F": { "strength": 0.13, "grain_size": 1.0, "color_var": 0.06, "sharpen": 0.0 }, + "CineStill_800T": { "strength": 0.2, "grain_size": 1.8, "color_var": 0.09, "sharpen": 0.1 }, + "Kodak_Ektachrome_100": { "strength": 0.10, "grain_size": 0.9, "color_var": 0.05, "sharpen": 0.0 }, + "Ilford_Delta_3200": { "strength": 0.32, "grain_size": 2.8, "color_var": 0.0, "sharpen": 0.15 }, + "Fujifilm_Terra": { "strength": 0.16, "grain_size": 1.3, "color_var": 0.07, "sharpen": 0.05 } +} diff --git a/colortools/ColorAnalysisPlotNode.py b/colortools/ColorAnalysisPlotNode.py new file mode 100644 index 0000000..dfd0e06 --- /dev/null +++ b/colortools/ColorAnalysisPlotNode.py @@ -0,0 +1,170 @@ +# © 2025 rndnanthu – Licensed under CC BY-NC 4.0 (https://creativecommons.org/licenses/by-nc/4.0/) +import torch +import numpy as np +import cv2 +import matplotlib.pyplot as plt +import io +from PIL import Image + +class ColorAnalysisPlotNode: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "image": ("IMAGE",), + "plot_type": (["histogram", "parade", "waveform", "vectorscope", "false_color", "gamut_warning"],), + "exposure": ("FLOAT", {"default": 1.0, "min": 0.1, "max": 5.0, "step": 0.1}), + } + } + + RETURN_TYPES = ("IMAGE",) + RETURN_NAMES = ("scope_image",) + FUNCTION = "analyze" + CATEGORY = "rndnanthu/🎨Color Tools" + + def analyze(self, image, plot_type="histogram", exposure=1.0): + if image.ndim != 4 or image.shape[0] != 1 or image.shape[-1] != 3: + raise ValueError(f"Expected image shape (1, H, W, 3), got {image.shape}") + # Always keep both formats ready + img_np = image[0].cpu().numpy() + img_np = np.clip(img_np * exposure, 0, 1) + img_torch = torch.clamp(image * exposure, 0, 1) + + if plot_type == "histogram": + result_np = self.plot_histogram(img_np) + elif plot_type == "parade": + result_torch = self.plot_parade(img_torch) + result_np = result_torch[0].cpu().numpy() + elif plot_type == "waveform": + result_torch = self.plot_waveform(img_torch) + result_np = result_torch[0].cpu().numpy() + elif plot_type == "vectorscope": + # 🟢 Use torch directly + result_torch = self.plot_vectorscope(img_torch) + result_np = result_torch[0].cpu().numpy() + elif plot_type == "false_color": + result_np = self.plot_false_color(img_np) + elif plot_type == "gamut_warning": + result_np = self.plot_gamut_warning(img_np) + else: + raise ValueError(f"Unknown plot type: {plot_type}") + + return (torch.from_numpy(result_np.astype(np.float32))[None, ...],) + + def plot_histogram(self, img): + h, w, _ = img.shape + fig, ax = plt.subplots(figsize=(6, 4), dpi=100) + for i, color in enumerate(['r', 'g', 'b']): + hist = cv2.calcHist([img.astype(np.float32)], [i], None, [256], [0, 1]) + ax.plot(hist, color=color, linewidth=1) + ax.set_xlim([0, 256]) + ax.set_title("RGB Histogram") + ax.axis('off') + return self._fig_to_img(fig, h, w) + + def plot_parade(self, img_tensor: torch.Tensor) -> torch.Tensor: + assert img_tensor.ndim == 4 and img_tensor.shape[0] == 1 and img_tensor.shape[3] == 3, \ + f"Expected shape (1, H, W, 3), got {img_tensor.shape}" + + device = img_tensor.device + B, H, W, C = img_tensor.shape + scope = torch.zeros((1, H, W * 3, 3), dtype=torch.float32, device=device) + + for ch in range(3): + channel = img_tensor[0, :, :, ch] + y_indices = ((1.0 - channel) * (H - 1)).round().long().clamp(0, H - 1) + x_indices = torch.arange(W, device=device).expand(H, W) + y_flat = y_indices.flatten() + x_flat = x_indices.flatten() + + acc = torch.zeros((H, W), dtype=torch.float32, device=device) + acc.index_put_((y_flat, x_flat), torch.ones_like(y_flat, dtype=torch.float32), accumulate=True) + + col_max = acc.max(dim=0, keepdim=True).values + col_max[col_max == 0] = 1 + acc = acc / col_max + acc = acc.pow(0.5) + + scope[0, :, ch * W:(ch + 1) * W, ch] = acc + + return scope + + def plot_waveform(self, img: torch.Tensor) -> torch.Tensor: + """ + RGB Waveform (torch, CUDA). + Expects img as (1, H, W, 3) or (H, W, 3) on CUDA, in range [0,1]. + Output: (1, H, W, 3) on CUDA + """ + if img.dim() == 4 and img.shape[0] == 1: + img = img[0] # (H, W, 3) + + H, W, _ = img.shape + scope = torch.zeros((H, W, 3), device=img.device) + + for c in range(3): + vals = img[..., c] + y_pos = ((1.0 - vals) * (H - 1)).long().clamp(0, H - 1) + x_range = torch.arange(W, device=img.device).repeat(H, 1) + y_range = y_pos + + scope[y_range, x_range, c] = 1.0 + + return scope.unsqueeze(0) + + def plot_vectorscope(self, img: torch.Tensor) -> torch.Tensor: + """ + Vectorscope plot using U/V chroma mapping. + Expects img as torch.Tensor (1, H, W, 3) on CUDA, in range [0,1]. + Output: (1, 512, 512, 3) on CUDA. + """ + img = img.squeeze(0) # (H, W, 3) + H, W, _ = img.shape + + rgb_to_yuv = torch.tensor([ + [0.299, -0.14713, 0.615], + [0.587, -0.28886, -0.51499], + [0.114, 0.436, -0.10001] + ], device=img.device) + + yuv = torch.tensordot(img, rgb_to_yuv, dims=([2], [0])) # (H, W, 3) + u = yuv[..., 1] + v = yuv[..., 2] + + px = ((u + 0.5) * 512).clamp(0, 511).long() + py = ((v + 0.5) * 512).clamp(0, 511).long() + + scope = torch.zeros((512, 512, 3), device=img.device) + scope[py, px] = torch.tensor([0.0, 1.0, 0.0], device=img.device) # Green points + + return scope.unsqueeze(0) + + + def plot_false_color(self, img): + gray = np.mean(img, axis=2) + color = cv2.applyColorMap((gray * 255).astype(np.uint8), cv2.COLORMAP_MAGMA) + return color[:, :, ::-1].astype(np.float32) / 255.0 # BGR → RGB + + def plot_gamut_warning(self, img): + warning = np.zeros_like(img) + over = np.any(img > 1.0, axis=-1) + under = np.any(img < 0.0, axis=-1) + warning[over] = [1, 0, 0] + warning[under] = [0, 0, 1] + return np.clip(img + warning * 0.5, 0, 1) + + def _fig_to_img(self, fig, target_h, target_w): + buf = io.BytesIO() + fig.savefig(buf, format='png', bbox_inches='tight', pad_inches=0) + buf.seek(0) + pil_img = Image.open(buf).convert("RGB") + pil_img = pil_img.resize((target_w, target_h)) + plt.close(fig) + return np.array(pil_img).astype(np.float32) / 255.0 + +NODE_CLASS_MAPPINGS = { + "ColorAnalysisPlotNode": ColorAnalysisPlotNode +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "ColorAnalysisPlotNode": "📈 Color Analysis Scope" +} diff --git a/colortools/ColorGradingNode.py b/colortools/ColorGradingNode.py new file mode 100644 index 0000000..5968b96 --- /dev/null +++ b/colortools/ColorGradingNode.py @@ -0,0 +1,124 @@ +# © 2025 rndnanthu – Licensed under CC BY-NC 4.0 +import torch +import numpy as np +import cv2 + +class ProColorGrading: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "image": ("IMAGE",), + + "shadows": ("FLOAT", {"default": 0.0, "min": -1.0, "max": 1.0, "step": 0.01}), + "midtones": ("FLOAT", {"default": 0.0, "min": -1.0, "max": 1.0, "step": 0.01}), + "highlights": ("FLOAT", {"default": 0.0, "min": -1.0, "max": 1.0, "step": 0.01}), + "exposure": ("FLOAT", {"default": 0.0, "min": -2.0, "max": 2.0, "step": 0.01}), + "hue_shift": ("FLOAT", {"default": 0.0, "min": -180.0, "max": 180.0, "step": 0.1}), + "saturation": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 3.0, "step": 0.01}), + "vibrance": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 3.0, "step": 0.01}), + "contrast": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 3.0, "step": 0.01}), + "pivot": ("FLOAT", {"default": 0.5, "min": 0.0, "max": 1.0, "step": 0.01}), + "color_boost": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 3.0, "step": 0.01}), + "temperature": ("FLOAT", {"default": 0.0, "min": -1.0, "max": 1.0, "step": 0.01}), + "tint": ("FLOAT", {"default": 0.0, "min": -1.0, "max": 1.0, "step": 0.01}), + "levels_in_black": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.01}), + "levels_in_white": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}), + "levels_out_black": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.01}), + "levels_out_white": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}), + } + } + + RETURN_TYPES = ("IMAGE",) + RETURN_NAMES = ("graded_image",) + FUNCTION = "grade" + CATEGORY = "rndnanthu/🎨Color Tools" + + def grade(self, image, shadows, midtones, highlights, + exposure, hue_shift, saturation, vibrance, + contrast, pivot, color_boost, temperature, tint, + levels_in_black, levels_in_white, levels_out_black, levels_out_white): + + img = image[0].cpu().numpy().astype(np.float32) + img = np.clip(img, 0, 1) + + # Exposure + if exposure != 0.0: + img *= 2.0 ** exposure + img = np.clip(img, 0, 1) + + # Input levels + if levels_in_black != 0.0 or levels_in_white != 1.0: + img = (img - levels_in_black) / max(1e-5, levels_in_white - levels_in_black) + img = np.clip(img, 0, 1) + + # Tone mask adjustments + if shadows != 0.0 or midtones != 0.0 or highlights != 0.0: + lum = np.mean(img, axis=2, keepdims=True) + shadows_mask = np.clip(1.0 - lum * 2.0, 0.0, 1.0) + highlights_mask = np.clip((lum - 0.5) * 2.0, 0.0, 1.0) + midtones_mask = 1.0 - shadows_mask - highlights_mask + + img += shadows * shadows_mask + img += midtones * midtones_mask + img += highlights * highlights_mask + img = np.clip(img, 0, 1) + + # Temperature/tint + if temperature != 0.0 or tint != 0.0: + wb = np.array([ + 1.0 + temperature * 0.1 - tint * 0.05, + 1.0, + 1.0 - temperature * 0.1 + tint * 0.05 + ], dtype=np.float32).reshape((1, 1, 3)) + img *= wb + img = np.clip(img, 0, 1) + + # HSV adjustments: only apply if needed + if hue_shift != 0.0 or saturation != 1.0 or vibrance != 0.0: + img_255 = (img * 255).astype(np.uint8) + hsv = cv2.cvtColor(img_255, cv2.COLOR_RGB2HSV) + h, s, v = cv2.split(hsv) + + if hue_shift != 0.0: + h = (h.astype(np.int32) + int(hue_shift)) % 180 + h = h.astype(np.uint8) + + if saturation != 1.0: + s = np.clip(s.astype(np.float32) * saturation, 0, 255).astype(np.uint8) + + if vibrance > 0.0: + mean_sat = np.mean(s) + mask = s < mean_sat + s[mask] = np.clip(s[mask] + vibrance * (255 - s[mask]), 0, 255).astype(np.uint8) + + hsv_mod = cv2.merge([h, s, v]) + img = cv2.cvtColor(hsv_mod, cv2.COLOR_HSV2RGB).astype(np.float32) / 255.0 + img = np.clip(img, 0, 1) + + # Contrast + if contrast != 1.0: + img = (img - pivot) * contrast + pivot + img = np.clip(img, 0, 1) + + # Color boost + if color_boost != 1.0: + avg = np.mean(img, axis=2, keepdims=True) + img = avg + (img - avg) * color_boost + img = np.clip(img, 0, 1) + + # Output levels + if levels_out_black != 0.0 or levels_out_white != 1.0: + img = img * (levels_out_white - levels_out_black) + levels_out_black + img = np.clip(img, 0, 1) + + return (torch.from_numpy(img).unsqueeze(0),) + + +NODE_CLASS_MAPPINGS = { + "ProColorGrading": ProColorGrading, +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "ProColorGrading": "🎛️ Pro Color Grading", +} diff --git a/colortools/autogradepro.py b/colortools/autogradepro.py new file mode 100644 index 0000000..1a87cd3 --- /dev/null +++ b/colortools/autogradepro.py @@ -0,0 +1,168 @@ +# © 2025 rndnanthu – Licensed under CC BY-NC 4.0 (https://creativecommons.org/licenses/by-nc/4.0/) +import torch +import numpy as np +import cv2 + +class AutoGradeProNode: + @classmethod + def INPUT_TYPES(cls): + return {"required": { + "image": ("IMAGE",), + "enable_white_balance": ("BOOLEAN", {"default": True}), + "enable_exposure": ("BOOLEAN", {"default": True}), + "enable_contrast": ("BOOLEAN", {"default": True}), + "enable_skintone_preserve": ("BOOLEAN", {"default": True}), + "enable_hdr_enhance": ("BOOLEAN", {"default": False}), + "wb_strength": ("FLOAT", {"default": 1.0, "min": 0, "max": 1, "step": 0.01}), + "exposure_strength": ("FLOAT", {"default": 1.0, "min": 0, "max": 1, "step": 0.01}), + "contrast_strength": ("FLOAT", {"default": 1.0, "min": 0, "max": 1, "step": 0.01}), + "skintone_strength": ("FLOAT", {"default": 1.0, "min": 0, "max": 1, "step": 0.01}), + "hdr_strength": ("FLOAT", {"default": 0.8, "min": 0, "max": 1, "step": 0.01}), + }} + + RETURN_TYPES = ("IMAGE",) + RETURN_NAMES = ("corrected_image",) + FUNCTION = "process" + CATEGORY = "rndnanthu/🎨Color Tools" + + def process(self, image, enable_white_balance, enable_exposure, enable_contrast, + enable_skintone_preserve, enable_hdr_enhance, + wb_strength, exposure_strength, contrast_strength, + skintone_strength, hdr_strength): + + tensor = image[0] + + # Convert to numpy image in [H, W, C] + if tensor.shape[0] == 3: # [C, H, W] + img = tensor.permute(1, 2, 0).cpu().numpy() # → [H, W, C] + elif tensor.shape[2] == 3: # already [H, W, C] + img = tensor.cpu().numpy() + else: + raise ValueError(f"Unsupported tensor shape: {tensor.shape}") + + img = self.ensure_rgb(img) + original = img.copy() + + if enable_white_balance: + img = self.white_balance_grayworld(img, wb_strength) + + if enable_exposure: + img = self.auto_exposure(img, exposure_strength) + + if enable_contrast: + img = self.auto_contrast(img, contrast_strength) + + if enable_skintone_preserve: + img = self.preserve_skintones(original, img, skintone_strength) + + if enable_hdr_enhance: + img = self.hdr_enhance(original, img, hdr_strength) + + # 🧠 Output in channel-last format: [1, H, W, C] + out = torch.from_numpy(np.clip(img, 0, 1)).unsqueeze(0).contiguous() + + return (out,) + + + + # === 🔧 Safety First — Enforce RGB Shape === + + def ensure_rgb(self, img): + """Force (H, W, 3) RGB format, regardless of grayscale, 1-channel, etc.""" + if img.ndim == 2: + img = np.stack([img] * 3, axis=-1) + elif img.ndim == 3: + if img.shape[2] == 1: + img = np.repeat(img, 3, axis=2) + elif img.shape[2] != 3: + raise ValueError(f"Invalid channel count: {img.shape}") + else: + raise ValueError(f"Image has invalid dimensions: {img.shape}") + return np.clip(img, 0, 1).astype(np.float32) + + + # === 🎨 Correction Functions === + + def white_balance_grayworld(self, img, strength): + img = self.ensure_rgb(img) + avg = np.mean(img, axis=(0, 1)) + gray = np.mean(avg) + gain = gray / (avg + 1e-6) + balanced = img * gain + return img * (1 - strength) + balanced * strength + + def auto_exposure(self, img, strength): + img = self.ensure_rgb(img) + + if img.ndim != 3 or img.shape[2] != 3: + print(f"[ERROR] auto_exposure: image shape is invalid: {img.shape}") + raise ValueError(f"Image must be RGB with 3 channels, got shape {img.shape}") + + # Force correct shape before OpenCV conversion + rgb_uint8 = (img * 255).astype(np.uint8) + if rgb_uint8.ndim != 3 or rgb_uint8.shape[2] != 3: + rgb_uint8 = np.stack([rgb_uint8] * 3, axis=-1) + + print(f"[DEBUG] rgb_uint8 shape: {rgb_uint8.shape}") # Must be (H, W, 3) + + # Convert to HSV + hsv = cv2.cvtColor(rgb_uint8, cv2.COLOR_RGB2HSV) + + # Work on the V channel + v = hsv[:, :, 2].astype(np.float32) / 255.0 + v_min, v_max = np.percentile(v, [2, 98]) + stretched = np.clip((v - v_min) / (v_max - v_min + 1e-6), 0, 1) + hsv[:, :, 2] = ((1 - strength) * v + strength * stretched) * 255 + + out_rgb = cv2.cvtColor(hsv.astype(np.uint8), cv2.COLOR_HSV2RGB) + return out_rgb.astype(np.float32) / 255.0 + + + def auto_contrast(self, img, strength): + img = self.ensure_rgb(img) + yuv = cv2.cvtColor((img * 255).astype(np.uint8), cv2.COLOR_RGB2YUV) + y = yuv[:, :, 0].astype(np.float32) / 255.0 + mean = y.mean() + stretched = 0.5 * (1 + np.tanh(2 * (y - mean))) + yuv[:, :, 0] = ((1 - strength) * y + strength * stretched) * 255 + return cv2.cvtColor(yuv.astype(np.uint8), cv2.COLOR_YUV2RGB).astype(np.float32) / 255.0 + + def preserve_skintones(self, original, corrected, strength): + original = self.ensure_rgb(original) + corrected = self.ensure_rgb(corrected) + + # Convert to HSV + hsv = cv2.cvtColor((original * 255).astype(np.uint8), cv2.COLOR_RGB2HSV) + + # Define skin tone range (in OpenCV HSV: H=0~180, S=0~255, V=0~255) + lower = np.array([0, 30, 60], dtype=np.uint8) # reddish hue, min saturation and brightness + upper = np.array([35, 180, 255], dtype=np.uint8) # yellow-orange hue, max saturation + + skin_mask = cv2.inRange(hsv, lower, upper).astype(np.float32) / 255.0 # [0.0, 1.0] float mask + + # Smooth the mask to avoid hard edges + skin_mask = cv2.GaussianBlur(skin_mask, (15, 15), 0) + + # Expand mask to match shape [H, W, 3] + mask_3c = np.repeat(skin_mask[:, :, np.newaxis], 3, axis=2) + + # Blend corrected and original using the skin mask + blended = corrected * (1 - mask_3c * strength) + original * (mask_3c * strength) + return blended + + + def hdr_enhance(self, original, corrected, strength): + corrected = self.ensure_rgb(corrected) + enhanced = cv2.detailEnhance((corrected * 255).astype(np.uint8), sigma_s=10, sigma_r=0.15) + enhanced = enhanced.astype(np.float32) / 255.0 + return corrected * (1 - strength) + enhanced * strength + + +# === Bind to ComfyUI === + +NODE_CLASS_MAPPINGS = { + "AutoGradePro": AutoGradeProNode +} +NODE_DISPLAY_NAME_MAPPINGS = { + "AutoGradePro": "🎨 AutoGradePro 2.0" +} diff --git a/colortools/colorspacesim.py b/colortools/colorspacesim.py new file mode 100644 index 0000000..a46f110 --- /dev/null +++ b/colortools/colorspacesim.py @@ -0,0 +1,162 @@ +# © 2025 rndnanthu – Licensed under CC BY-NC 4.0 (https://creativecommons.org/licenses/by-nc/4.0/) +import torch +import numpy as np +import os +import glob + +BASE_PATH = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) +LUT_FOLDER = os.path.join(BASE_PATH, "assets", "luts") + +class ColorSpaceSimNode: + @classmethod + def INPUT_TYPES(cls): + lut_files = glob.glob(os.path.join(LUT_FOLDER, "*.cube")) + lut_names = [os.path.basename(f) for f in lut_files] + lut_map = dict(zip(lut_names, lut_files)) + cls._lut_map = lut_map + + return { + "required": { + "image": ("IMAGE",), + "target_profile": ( + ["Sony S-Log3", "Arri LogC", "Canon C-Log", "Rec.709"], + {"default": "Sony S-Log3"} + ), + "simulate_camera_encode": ("BOOLEAN", {"default": False}), + "enable_gamma": ("BOOLEAN", {"default": False}), + "gamma_value": ("FLOAT", {"default": 2.2, "min": 0.1, "max": 5.0, "step": 0.01}), + "enable_lut": ("BOOLEAN", {"default": False}), + "lut_name": ( + ["None"] + sorted(lut_names) if lut_names else ["None"], + {"default": "None"} + ), + "profile_strength": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}), + "enable_tone_map": ("BOOLEAN", {"default": True}), + "contrast_boost": ("FLOAT", {"default": 1.05, "min": 0.5, "max": 2.0, "step": 0.01}), + } + } + + RETURN_TYPES = ("IMAGE",) + RETURN_NAMES = ("out",) + FUNCTION = "run" + CATEGORY = "rndnanthu/🎨Color Tools" + + def run(self, image, target_profile, simulate_camera_encode, + enable_gamma, gamma_value, enable_lut, lut_name, + profile_strength, enable_tone_map, contrast_boost): + + img = image # shape (1, H, W, 3) + if img.ndim != 4 or img.shape[-1] != 3: + raise ValueError(f"Expected image shape (1, H, W, 3), got {img.shape}") + + img = img.clamp(0, 1).float() + original = img.clone() + + if simulate_camera_encode and target_profile != "Rec.709": + img = self.srgb_to_linear_tensor(img) + + if target_profile == "Sony S-Log3": + img = self.slog3_curve_tensor(img) + elif target_profile == "Arri LogC": + img = self.logc_curve_tensor(img) + elif target_profile == "Canon C-Log": + img = self.clog_curve_tensor(img) + + if enable_tone_map: + img = self.simple_tone_map_tensor(img) + + if enable_gamma: + img = torch.pow(img.clamp(0, 1), 1.0 / gamma_value) + + if enable_lut and lut_name in self._lut_map: + try: + img = self.apply_cube_lut_gpu(img, self._lut_map[lut_name]) + except Exception as e: + print("⚠️ LUT failed:", e) + + if contrast_boost != 1.0: + img = self.boost_contrast_tensor(img, contrast_boost) + + final = torch.clamp(original * (1.0 - profile_strength) + img * profile_strength, 0, 1) + return (final,) + + def srgb_to_linear_tensor(self, img): + return torch.where( + img <= 0.04045, img / 12.92, + torch.pow((img + 0.055) / 1.055, 2.4) + ) + + def slog3_curve_tensor(self, x): + a, b = 0.432699, 0.037584 + return torch.where( + x < 0.011, + 0.092864 * x + 0.00404, + a * torch.log10(x + b) + 0.616596 + ) + + def logc_curve_tensor(self, x): + a, b, c, d, e, f, cut = 5.555556, 0.052272, 0.24719, 0.385537, 5.367655, 0.092809, 0.010591 + return torch.where( + x > cut, + c * torch.log10(a * x + b) + d, + e * x + f + ) + + def clog_curve_tensor(self, x): + return torch.log10(500 * x + 1) / np.log10(501) + + def simple_tone_map_tensor(self, x): + return x / (x + 0.25) + + def boost_contrast_tensor(self, x, boost): + mid = 0.5 + return torch.clamp((x - mid) * boost + mid, 0, 1) + + def load_cube_lut_torch(self, path): + with open(path, 'r') as f: + lines = [l.strip() for l in f if l.strip() and not l.startswith('#')] + + size_line = [l for l in lines if l.startswith('LUT_3D_SIZE')] + if not size_line: + raise ValueError("Missing LUT_3D_SIZE in cube file") + + size = int(size_line[0].split()[1]) + + data = [list(map(float, l.split())) for l in lines if len(l.split()) == 3] + if len(data) != size ** 3: + raise ValueError(f"Invalid LUT size: expected {size**3}, got {len(data)}") + + lut_np = np.array(data, dtype=np.float32).reshape((size, size, size, 3)) # (R, G, B, 3) + lut_torch = torch.from_numpy(lut_np).permute(3, 0, 1, 2).contiguous() # (3, R, G, B) + return lut_torch, size + + def apply_cube_lut_gpu(self, img, path): + lut, size = self.load_cube_lut_torch(path) + + if not torch.is_tensor(img) or img.ndim != 4 or img.shape[-1] != 3: + raise ValueError("Expected image shape (1, H, W, 3)") + + device = img.device + B, H, W, C = img.shape + + # normalize image to [-1, 1] + norm = img * 2 - 1 + grid = norm.view(B, H, W, 1, C) # (B, H, W, 1, 3) + + lut = lut.to(device).unsqueeze(0) # shape: (1, 3, size, size, size) + + sampled = torch.nn.functional.grid_sample( + lut, grid, mode='bilinear', align_corners=True + ) # returns (1, 3, H, W, 1) + + out = sampled.squeeze(-1).permute(0, 2, 3, 1) # -> (1, H, W, 3) + return out.clamp(0, 1) + + +NODE_CLASS_MAPPINGS = { + "ColorSpaceSim": ColorSpaceSimNode +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "ColorSpaceSim": "🎥 Color Space Simulator" +} diff --git a/colortools/log_color_conversion.py b/colortools/log_color_conversion.py new file mode 100644 index 0000000..542b3e1 --- /dev/null +++ b/colortools/log_color_conversion.py @@ -0,0 +1,173 @@ +# © 2025 rndnanthu – Licensed under CC BY-NC 4.0 (https://creativecommons.org/licenses/by-nc/4.0/) +import os +import torch +import numpy as np +import cv2 + +# Optional EXR support +try: + import OpenEXR, Imath + exr_available = True +except ImportError: + exr_available = False + + +class ConvertToLogImage: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "image": ("IMAGE",), + "log_type": (["cineon", "logc", "redlog"], {"default": "cineon"}), + "is_srgb": ("BOOLEAN", {"default": True}), + "tone_map_hdr": ("BOOLEAN", {"default": False}), + "enhance_bit_depth": ("BOOLEAN", {"default": False}), + "save_as_openexr": ("BOOLEAN", {"default": False}), + "exr_filename": ("STRING", {"default": "log_output.exr"}), + }, + "optional": { + "depth_map": ("IMAGE",), + "gamma_value": ("FLOAT", {"default": 1.0, "min": 0.5, "max": 2.0, "step": 0.05}), + "contrast": ("FLOAT", {"default": 1.8, "min": 0.1, "max": 5.0, "step": 0.01}), + "midpoint": ("FLOAT", {"default": 0.5, "min": 0.0, "max": 1.0, "step": 0.01}), + "depth_boost": ("FLOAT", {"default": 0.20, "min": 0.0, "max": 1.0, "step": 0.01}), + } + } + + RETURN_TYPES = ("IMAGE",) + RETURN_NAMES = ("log_image",) + FUNCTION = "run" + CATEGORY = "rndnanthu/🎨Color Tools" + + # === Color Conversion === + + def srgb_to_linear(self, c): + return np.where(c <= 0.04045, c / 12.92, ((c + 0.055) / 1.055) ** 2.4) + + def rec709_to_linear(self, c): + return np.where(c < 0.081, c / 4.5, ((c + 0.099) / 1.099) ** (1 / 0.45)) + + def tone_map(self, l): + return l / (l + 1.0) + + def linear_to_cineon(self, l): + l = np.clip(l, 1e-6, 1) + return (np.log10(l * 685 + 95) - np.log10(95)) / (np.log10(780) - np.log10(95)) + + def linear_to_logc(self, l): + a, b, c_, d, e, f, cut = 5.555556, 0.052272, 0.2471896, 0.385537, 5.367655, 0.092809, 0.010591 + l = np.clip(l, 1e-6, 1) + return np.where(l > cut, c_ * np.log10(a * l + b) + d, e * l + f) + + def linear_to_redlog(self, l): + l = np.clip(l, 1e-6, 1) + return (np.log10(l + 0.01) + 0.6) / 1.6 + + def convert_to_log(self, l, log_type): + funcs = { + "cineon": self.linear_to_cineon, + "logc": self.linear_to_logc, + "redlog": self.linear_to_redlog + } + return funcs[log_type](l) + + # === Dynamic Range Enhancement === + + def enhance_dynamic_range(self, img, gamma_value, contrast, midpoint, depth_map=None, depth_boost=0.5): + img = np.clip(img, 0.001, 1.0) + + # === Step 1: Depth boost applied FIRST in linear space === + if depth_map is not None: + dmap = depth_map.squeeze().cpu().numpy().astype(np.float32) + + # If 3-channel, convert to grayscale + if dmap.ndim == 3 and dmap.shape[-1] == 3: + dmap = cv2.cvtColor(dmap, cv2.COLOR_RGB2GRAY) + + # Resize & blur + dmap = cv2.resize(dmap, (img.shape[1], img.shape[0]), interpolation=cv2.INTER_LINEAR) + dmap = cv2.GaussianBlur(dmap, (5, 5), 0) + dmap = np.clip(dmap, 0.0, 1.0) + + # Expand to RGB shape + dmap = np.expand_dims(dmap, axis=-1) + dmap = np.repeat(dmap, 3, axis=2) + + # Apply depth boost + img += dmap * depth_boost + img = np.clip(img, 0.001, 1.0) # Re-clip after depth influence + + # === Step 2: Gamma correction === + img_gamma = np.power(img, gamma_value) + + # === Step 3: Sigmoid contrast === + img_contrast = 1 / (1 + np.exp(-contrast * (img_gamma - midpoint))) + + return np.clip(img_contrast, 0, 1) + + + # === EXR Export === + + def export_exr(self, img_np, path): + if not exr_available: + print("[Log Converter] OpenEXR not available.") + return + if img_np.shape[-1] != 3: + raise ValueError("EXR export requires 3-channel RGB image.") + h, w, _ = img_np.shape + header = OpenEXR.Header(w, h) + pt = Imath.PixelType(Imath.PixelType.FLOAT) + header['channels'] = {'R': Imath.Channel(pt), 'G': Imath.Channel(pt), 'B': Imath.Channel(pt)} + exr = OpenEXR.OutputFile(path, header) + r, g, b = [img_np[:, :, i].astype(np.float32).tobytes() for i in range(3)] + exr.writePixels({'R': r, 'G': g, 'B': b}) + exr.close() + + # === Main === + + def run( + self, image, log_type, is_srgb, tone_map_hdr, enhance_bit_depth, + save_as_openexr, exr_filename, depth_map=None, + gamma_value=0.8, contrast=1.2, midpoint=0.5, depth_boost=0.5 + ): + if not isinstance(image, torch.Tensor) or image.ndim != 4 or image.shape[-1] != 3: + raise ValueError(f"[Log Converter] Expected shape [B,H,W,3], got: {image.shape}") + + img = image[0].cpu().numpy() # [H,W,3] + img = np.clip(img, 0, 1).astype(np.float32) + + # Convert to linear + linear = self.srgb_to_linear(img) if is_srgb else self.rec709_to_linear(img) + + # Tone map if needed + if tone_map_hdr: + linear = self.tone_map(linear) + + # Convert to log + log_img = self.convert_to_log(linear, log_type) + log_img = np.clip(log_img, 0, 1) + + # Bit-depth enhancement + if enhance_bit_depth: + log_img = self.enhance_dynamic_range( + log_img, gamma_value, contrast, midpoint, depth_map, depth_boost + ) + + # Save as EXR + if save_as_openexr: + os.makedirs("output", exist_ok=True) + self.export_exr(log_img, os.path.join("output", exr_filename)) + + output_tensor = torch.from_numpy(log_img).unsqueeze(0).contiguous().float() + return (output_tensor,) + + +# === ComfyUI Registration === + +NODE_CLASS_MAPPINGS = { + "ConvertToLogImage": ConvertToLogImage +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "ConvertToLogImage": "🎚️ Log Converter (Cineon/LogC/REDLog)" +} diff --git a/grain/FilmGrain.py b/grain/FilmGrain.py new file mode 100644 index 0000000..4cd67ba --- /dev/null +++ b/grain/FilmGrain.py @@ -0,0 +1,102 @@ +# © 2025 rndnanthu – Licensed under CC BY-NC 4.0 (https://creativecommons.org/licenses/by-nc/4.0/) +import torch +import numpy as np +import cv2 +import os +import json + +BASE_PATH = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) +PRESETS_FILE = os.path.join(BASE_PATH, "assets", "presets", "film_grain_presets.json") + +class FilmGrainNode: + @classmethod + def INPUT_TYPES(cls): + presets = [] + if os.path.exists(PRESETS_FILE): + try: + with open(PRESETS_FILE, "r") as f: + presets = list(json.load(f).keys()) + except Exception as e: + print("Error loading presets:", e) + return { + "required": { + "image": ("IMAGE",), + "preset_name": (["None"] + presets, {"default": "None"}), + "intensity": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 2.0, "step": 0.01}), + "strength": ("FLOAT", {"default": 0.2, "min": 0.0, "max": 1.0, "step": 0.01} ), + "grain_size": ("FLOAT", {"default": 1.2, "min": 0.5, "max": 3.0, "step": 0.01}), + "color_var": ("FLOAT", {"default": 0.1, "min": 0.0, "max": 1.0, "step": 0.01}), + "sharpen": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 2.0, "step": 0.01}), + } + } + + RETURN_TYPES = ("IMAGE",) + RETURN_NAMES = ("grained",) + FUNCTION = "apply_grain" + CATEGORY = "rndnanthu/🎨Grain Tools" + + def __init__(self): + self.presets = {} + self.load_presets() + + def load_presets(self): + if os.path.exists(PRESETS_FILE): + try: + with open(PRESETS_FILE, "r") as f: + self.presets = json.load(f) + except Exception as e: + print("Failed to load presets:", e) + self.presets = {} + + def apply_grain(self, image, preset_name, intensity, strength, grain_size, color_var, sharpen): + img = image[0].cpu().numpy() # (H, W, C) + + # Apply preset if selected + if preset_name and preset_name in self.presets: + p = self.presets[preset_name] + strength = p.get("strength", strength) + grain_size = p.get("grain_size", grain_size) + color_var = p.get("color_var", color_var) + sharpen = p.get("sharpen", sharpen) + + # Apply intensity globally + strength *= intensity + color_var *= intensity + sharpen *= intensity + + h, w, c = img.shape # (H, W, 3) + + # === Generate monochromatic noise === + mono_noise = np.random.normal(0, 1, (h, w)).astype(np.float32) + mono_noise = cv2.GaussianBlur(mono_noise, (0, 0), grain_size) + mono_noise = ((mono_noise - mono_noise.min()) / (mono_noise.max() - mono_noise.min() + 1e-8)) * 2 - 1 + mono_noise = np.repeat(mono_noise[:, :, np.newaxis], 3, axis=2) + + # === Generate chromatic noise === + chroma_noise = np.random.normal(0, 1, (h, w, 3)).astype(np.float32) + for ch in range(3): + chroma_noise[:, :, ch] = cv2.GaussianBlur(chroma_noise[:, :, ch], (0, 0), grain_size) + chroma_noise = ((chroma_noise - chroma_noise.min()) / (chroma_noise.max() - chroma_noise.min() + 1e-8)) * 2 - 1 + + # === Combine noise === + grain = (1.0 - color_var) * mono_noise + color_var * chroma_noise + grained = img + grain * strength + grained = np.clip(grained, 0.0, 1.0) + + # === Optional sharpening === + if sharpen > 0.0: + blur = cv2.GaussianBlur(grained, (0, 0), 1.0) + grained = cv2.addWeighted(grained, 1 + sharpen, blur, -sharpen, 0) + + output = torch.from_numpy(grained).unsqueeze(0).float() # (1, H, W, 3) + return (output,) + + + +NODE_CLASS_MAPPINGS = { + "FilmGrain": FilmGrainNode +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "FilmGrain": "🎞 Film Grain (Advanced)" +} diff --git a/prompt/PromptGenerator.py b/prompt/PromptGenerator.py new file mode 100644 index 0000000..5a13a4e --- /dev/null +++ b/prompt/PromptGenerator.py @@ -0,0 +1,370 @@ +# © 2025 rndnanthu – Licensed under CC BY-NC 4.0 (https://creativecommons.org/licenses/by-nc/4.0/) +import base64 +import io +import requests +import numpy as np +from PIL import Image + +class LMStudioMultimodalPrompt: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "mode_selection": (["image", "video", "logo"], { + "default": "video" + }), + "lm_api_url": ("STRING", { + "default": "http://localhost:1234/v1/chat/completions" + }), + "model_name": ("STRING", { + "default": "lmstudio-community/llava-phi-3" + }), + "prompt_input": ("STRING", { + "multiline": True, + "default": "a mysterious hooded figure walks through snowfall at night" + }), + "instruction_input": ("STRING", { + "multiline": True, + "default": "Use handheld camera and moody lighting" + }), + }, + "optional": { + "image_input": ("IMAGE",), + } + } + + RETURN_TYPES = ("STRING",) + RETURN_NAMES = ("generated_prompt",) + FUNCTION = "generate" + CATEGORY = "rndnanthu/prompt" + + # === Image Tensor Helper === + def tensor_to_base64(self, image_tensor): + image_array = (image_tensor.cpu().numpy() * 255).clip(0, 255).astype(np.uint8) + image = Image.fromarray(image_array) + buf = io.BytesIO() + image.save(buf, format='PNG') + return base64.b64encode(buf.getvalue()).decode("utf-8") + + # === System Prompts === + + def get_text_to_video_system_prompt(self): + return """You are a high-precision multimodal prompt generator specialized in producing cinematic, anatomically detailed video scene descriptions. Your primary task is to convert user-provided text prompts into vivid, camera-ready descriptions suitable for driving realistic 5–10 second video generations. + + 🎯 PRIMARY OBJECTIVE: + Transform a simple, unstructured user prompt into a highly visual, emotionally resonant, and physically grounded video scene. Your descriptions must feel like a live camera feed capturing a meticulously staged, cinematic moment. You must describe **everything that would be visible inside the frame**: subject anatomy, wardrobe, facial expression, pose, motion, lighting, camera behavior, environment, and spatial composition. + + 🧩 INPUT STRUCTURE: + You may be given one or both of the following: + 1. A **plain video prompt** (e.g., “girl running on beach at sunset”). + 2. An optional **instruction block**, which may contain overrides, stylistic preferences, emotional mood, camera or lens specifications, or clarifying direction. + + ✅ INSTRUCTION PRIORITY: + - Treat instruction blocks as **authoritative directives**. If there is a conflict between the prompt and instructions, the **instruction always overrides the base prompt**. + - Never ignore instructions — follow all directives exactly unless they introduce physical or visual impossibilities. + - Examples: + - If prompt says “woman in red dress” but instruction says “torn jeans and a hoodie,” obey the hoodie. + - If prompt has no camera info but instruction says “tracking tilt-up with 35mm lens,” honor that framing. + + 🧠 STYLE & LANGUAGE: + - Use vivid, emotionally precise, hyper-realistic language. + - Never use vague descriptors like “beautiful,” “cool,” or “awesome.” Instead, use **observable, grounded physical descriptions** to convey visual quality. + - Never summarize or abstract — **focus entirely on describing what the camera sees inside the shot**. + + 📷 STRUCTURAL OUTPUT: (Always a flowing single paragraph) + Include the following visual components with clarity and precision: + + 1. **Subject**: + - Describe the subject’s gender, age, ethnicity, build, hairstyle, facial structure, posture, limbs, hands, fingers — every visible anatomical detail. + 2. **Clothing**: + - Detail garments by type, texture, fabric, fit, layering, wear, motion, and visible accessories (e.g., necklace, shoes). + 3. **Pose & Gesture**: + - Explicitly describe what every limb and body part is doing, and how weight is distributed. + 4. **Facial Expression**: + - Highlight specific muscle positions, gaze, eye intensity, lip tension, emotional signals from jaw, brows, and cheek motion. + 5. **Environment**: + - Describe physical location, objects, background, terrain, time of day, weather, light interaction, and context-defining elements. + 6. **Lighting**: + - Precisely define light sources, color temperature, shadow direction, bounce light, rim light, atmospheric haze, specular detail. + 7. **Camera Framing**: + - Include shot type (close-up, wide, medium), angle (low, high, eye level), lens (e.g., 35mm), framing, and any camera motion (e.g., dolly, pan, tilt). + 8. **Motion & Timing**: + - Describe what action is occurring live in the frame: body twists, hair caught mid-air, fabric movement — it must feel like an exact cinematic moment frozen in flow. + 9. **Fantasy/Sci-Fi Elements**: + - Only include speculative or fantastical features (e.g., magic, cybernetics, levitation) if the **instruction explicitly demands it**. + + 🚫 ABSOLUTE RULES: + - Always describe exactly **one human subject only**. Never include additional figures, reflections, shadows-as-characters, or imaginary beings. + - No summarizing, listing, or vague interpretation. + - Never generate visual hallucinations — all descriptions must follow physical realism unless otherwise directed. + + 🛑 PRESENTATION FORMAT: + - A single flowing paragraph, written in **present tense**, as if describing a shot playing on a director’s monitor. + - Never break format, never revert to list style. + - Your goal is to deliver a visually stunning, photorealistic scene description that reads like it’s unfolding in real time under a cinema lens. + """ + + + def get_image_to_video_system_prompt(self): + return """You are a multimodal image-to-video prompt engine designed to expand a single frame into a vivid, physically accurate, cinematic video scene. Your core function is to convert a **single-subject input image**, combined with optional instructions, into a 5-second continuous video description — maintaining all visual realism and emotional consistency. + + 🎯 OBJECTIVE: + Create a high-fidelity cinematic moment that continues the visual truth of the given image. Expand the frame into a living shot without deviating from what is visible or physically implied. Every part of your output must reflect what is grounded in the image, augmented only by instruction block if provided. + + 📸 INPUT FORMAT: + - One image input, featuring **exactly one subject**. + - Optionally, an **instruction block** that may include: + - Pose modification + - Camera lens and framing + - Mood or emotional tone + - Motion direction + - Outfit alterations + + ✅ INSTRUCTION RULES: + - Treat instruction block as authoritative. If instructions suggest different pose, attire, or lens, obey them. + - Only override what is visible in the image if instruction explicitly requests it. + - Never invent new subjects, additional people, or impossible motion/lighting. + + 🧠 IMAGE-GROUNDED LOGIC: + You must preserve all visible aspects of the image: + - Respect current pose, camera position, and lighting unless clearly overridden. + - All additions must be plausible within the physical scene visible. + - If posture or gaze is ambiguous, infer only **realistic cinematic outcomes** based on body language. + + 🖼️ STRUCTURE OF OUTPUT: + Write a single flowing paragraph including: + + 1. **Subject identity**: + - Gender, age, ethnicity, build, hairstyle, skin tone, face shape, and distinct features. + 2. **Anatomy & Pose**: + - Arms, hands, spine, shoulders, hips, leg tension, finger positioning — every visible posture detail. + 3. **Clothing**: + - Textures, folds, color, layers, worn condition, accessories, and motion of fabric. + 4. **Facial Expression**: + - Gaze, brow, cheek, lip and eye shape to infer mood. + 5. **Action & Motion**: + - Describe movement happening now or about to unfold — “left arm swings forward,” “heel lifts off surface,” etc. + 6. **Camera Work**: + - Shot framing, lens, movement (e.g. tracking, handheld), distance, and focus depth. + 7. **Environment**: + - Background details, light particles, fog, time-of-day cues, surface texture, wind. + 8. **Lighting**: + - Directional sources, backlight, mood contrast, highlights on skin or fabric, cinematic lighting ratios. + + 🚫 HARD RULES: + - The scene may contain **only one visible subject** — no additions, reflections as people, silhouettes, or hallucinated characters. + - Never speculate about parts of the image you cannot see — only infer from visible elements. + + 🛑 FORMAT: + - Return a **single paragraph**, written in **present tense**, as if you’re watching the expanded video live. + - Never use bullet points. No summaries or generalizations. + - Start with: “This video...” and continue with a flowing description of the moment. + + 📽️ Your job is to **extend the image into a cinematic moment** with perfect anatomical realism, expressive emotional cues, and plausible physical action. + """ + + + def get_text_to_image_system_prompt(self): + return """You are a high-fidelity image prompt generator. Your role is to take short user prompts and generate a rich, photorealistic, cinematic image description optimized for still frame generation. + + 🎯 OBJECTIVE: + Transform brief text prompts into a detailed, realistic, and emotionally resonant visual scene. Your description should feel like a high-resolution movie still — complete with subject anatomy, pose, lighting, environment, and lens-aware framing. + + 📷 OUTPUT STRUCTURE: + Respond in **one cinematic paragraph**, describing all of the following in vivid, physically grounded prose: + + 1. **Subject Identity**: + - Gender, age, ethnicity, body type, hair style, facial geometry, skin tone, scars, makeup, tattoos. + 2. **Pose & Anatomy**: + - Exact body configuration: posture, limbs, joints, hand gestures, and finger articulation. + 3. **Clothing**: + - Style, fabric, color, layering, accessories, wear/damage, motion (e.g. “jacket lifts slightly in wind”). + 4. **Facial Expression**: + - Gaze, mouth tension, eyebrow shape, emotional cues through muscle micro-movement. + 5. **Implied Motion**: + - Any frozen or about-to-happen gesture — "about to raise hand", "knee slightly bent". + 6. **Camera View**: + - Lens (macro, wide, shallow DOF), angle (low, top-down, eye-level), focus, and subject placement. + 7. **Environment**: + - Time of day, weather, background terrain, architectural elements, debris, shadows. + 8. **Lighting**: + - Directional light, softness, bounce, color temperature, rim light, volumetric haze or highlights. + + 🚫 RULES: + - Only describe one subject unless otherwise instructed. + - Do not invent surreal elements unless explicitly asked. + - Avoid summarizing or being vague — describe **what a camera sees**, not what a story tells. + + 🛑 FORMAT: + - Use present tense only. + - Write a **single, flowing, cinematic paragraph**. No bullet points or segmented lists. + """ + + def get_image_to_image_system_prompt(self): + return """You are a high-fidelity image-to-image interpreter designed to extract maximum cinematic detail and visual realism from a given image input. Your role is to describe the frame exactly as it appears — with no hallucination — using grounded, director-level vocabulary. Your output is meant to guide visual transformation, augmentation, or continuation while honoring what’s actually visible in the source image. + + 🎯 PRIMARY GOAL: + Transform the image into a highly accurate, frame-perfect description that captures everything visible within the current shot. Think like a cinematographer briefing an artist or AI: describe anatomical precision, fabric texture, camera logic, environmental cues, and lighting dynamics — always rooted in what is visible. + + 🧩 INPUT FORMAT: + - You will be provided one image. + - You may also receive a short prompt or instruction to guide interpretation. + - Never override or alter what is **not explicitly shown** unless instructed to. + + 📷 VISUAL ANALYSIS STRUCTURE: + Respond in a single, flowing paragraph, describing: + + 1. **Subject Identity**: + - Apparent gender, age, ethnicity, physique, hairstyle (including motion or texture), skin tone, face geometry, facial hair, markings or scars. + 2. **Anatomical Structure**: + - Full-body or partial-body pose: shoulders, spine alignment, hands/fingers, hips, knee bend, feet — describe every joint and gesture visible. + 3. **Clothing Detail**: + - Type of clothing, color, texture, fabric, fit, layering, motion (e.g., “cotton sleeve wrinkles at elbow”), and any accessories. + 4. **Facial Expression**: + - Eyebrow angle, brow tension, gaze direction, mouth curvature, tension in cheeks or jaw, blink, or half-closed eyes. Use subtle muscular cues to imply emotional tone. + 5. **Implied Motion**: + - What is happening in the moment? “Torso slightly rotated,” “hand halfway lifted,” “hair caught mid-turn” — freeze a believable action frame. + 6. **Camera Characteristics**: + - Shot angle, height, lens effect (depth of field, distortion, bokeh), focal distance, framing. Include any visible cinematic techniques (e.g., low-angle handheld framing). + 7. **Environment**: + - Ground material, ambient details, objects in background, time of day, weather particles, horizon line, or structural context (walls, trees, skyline). + 8. **Lighting**: + - Describe all light sources: warm, cold, rim light, bounce, fog light diffusion, reflective surfaces, and how the light interacts with skin or clothing. + + 🚫 STRICT CONSTRAINTS: + - You must never invent or hallucinate people, animals, extra limbs, shadows-as-persons, or reflections as second characters. + - Do not alter the camera angle or lighting unless the instruction explicitly tells you to. + - Maintain photorealism and anatomical logic — no distortions, no surreal transformations unless directed. + + 🛑 FORMAT: + - Write in **present tense**, as if you’re describing a paused high-resolution video frame to a director. + - Use a **single paragraph** with flowing, natural cinematic language. + - Avoid all lists, technical summaries, or categorical formatting — always write like a film set scene description. + """ + + + def get_text_to_logo_system_prompt(self): + return """You are a precision-focused brand identity prompt generator. Your job is to translate a short text-based brand or concept input into a complete, high-quality logo description — designed to inform and inspire a professional graphic designer. Your focus is on conveying strong symbolism, layout logic, and emotional branding cues through design language. + + 🎯 PRIMARY OBJECTIVE: + Transform a simple brand name or concept into a clean, iconic logo vision. Your description should include emblem shape, visual tone, layout style, typography cues, and color harmony — all in a way that aligns with the intended identity and communicates its emotion or purpose at a glance. + + 💡 VISUAL IDENTITY STRUCTURE TO INCLUDE: + Respond with a single flowing paragraph that includes: + + 1. **Iconography**: + - Describe symbolic representation: abstract shapes, literal icons, metaphoric visuals. Ensure they relate directly to the brand’s meaning. + 2. **Typography**: + - Font family tone (e.g., serif, sans-serif), casing (uppercase, mixed case), spacing, kerning, weight, and letter mood (modern, vintage, soft, sharp, geometric, elegant). + 3. **Layout**: + - Explain visual structure: horizontal, vertical, stacked, center-aligned, text-to-symbol ratio, margin logic, and symmetry or asymmetry. + 4. **Color Palette**: + - Suggest a palette that conveys the brand’s emotional mood: bold primaries, pastel gradients, monochrome minimalism, high-contrast neon, earth tones, etc. + 5. **Shape & Geometry**: + - Circle vs square vs custom, fluid lines vs angles, sharp corners vs curves — match geometry to theme (e.g., trust, speed, elegance, rebellion). + 6. **Design Style**: + - Overall visual tone: minimalistic, futuristic, nostalgic, luxury, eco-friendly, playful, edgy, or tech-centric. + + 🚫 CONSTRAINTS: + - Do not include scenes, landscapes, people, realistic faces, or photorealistic visuals. + - The design must remain 2D, graphic, and vector-style — **no 3D**, no gradients unless stylistic, no cinematic effects. + + 🛑 FORMAT: + - Write a **single present-tense paragraph** as if you’re briefing a senior logo designer or a generative AI logo model. + - Avoid bullet points. Use elegant brand vocabulary to convey visual structure, emotion, and balance. + """ + + def get_image_to_logo_system_prompt(self): + return """You are a logo analysis and refinement engine. Your job is to analyze an existing logo — either as an image or described prompt — and convert it into a structured branding brief using precise, professional design language. Your focus is on evaluating the logo’s layout, typography, symbol design, balance, and tone with clarity and objectivity. + + 🎯 GOAL: + Translate the visual style of the logo into a clear, detailed design system brief suitable for enhancement, iteration, or generative replication. Your output should capture the underlying visual principles of the logo — layout mechanics, icon proportions, typographic tone, and emotional feel — using designer vocabulary. + + 🖼️ VISUAL FEATURES TO DESCRIBE: + Respond with a clean, flowing paragraph that includes: + + 1. **Layout**: + - How is the icon arranged relative to text? (stacked, side-by-side, centered, left-justified). Describe spatial padding, proportions, and spacing. + 2. **Icon Design**: + - Detail the shapes and lines that define the symbol — abstract geometry, literal shapes, smooth curves, sharp angles, thickness, negative space, and visual symmetry. + 3. **Typography**: + - Analyze font type, weight, casing (upper/lower/mixed), alignment, character spacing, and emotional tone (friendly, futuristic, corporate, handcrafted, etc). + 4. **Color Scheme**: + - Describe dominant color usage: solids, gradients, black-and-white, muted tones, high-saturation pops, and contrast relationships. + 5. **Visual Tone**: + - Interpret the aesthetic feel — does it read as elegant, bold, minimal, retro, youthful, or corporate? + 6. **Symmetry & Balance**: + - Evaluate axis alignment, weight distribution, white space, and proportional flow across logo components. + + 🚫 LIMITATIONS: + - Never hallucinate components not visible in the image or prompt. + - Do not refer to backgrounds, lighting, realism, or 3D texture. + - Avoid describing any human subject or scene unless the logo icon itself includes stylized silhouettes or symbols. + + 🛑 FORMAT: + - Write in **present tense**, as if analyzing a logo during a design review. + - Return a **single paragraph** rich in design language, clear enough for a designer or model to fully reconstruct or modify the brand identity. + """ + + + # === Dispatcher === + def resolve_system_prompt(self, mode, is_vision): + if mode == "video": + return self.get_image_to_video_system_prompt() if is_vision else self.get_text_to_video_system_prompt() + elif mode == "image": + return self.get_image_to_image_system_prompt() if is_vision else self.get_text_to_image_system_prompt() + elif mode == "logo": + return self.get_image_to_logo_system_prompt() if is_vision else self.get_text_to_logo_system_prompt() + return "You are a visual prompt generator." + + # === Core Function === + def generate(self, mode_selection, lm_api_url, model_name, prompt_input, instruction_input, image_input=None): + is_vision = image_input is not None and len(image_input) > 0 + system_prompt = self.resolve_system_prompt(mode_selection, is_vision) + + messages = [{"role": "system", "content": system_prompt}] + + # Add instruction if present + if instruction_input.strip(): + messages.append({"role": "user", "content": instruction_input.strip()}) + + # Add image + text content if image is provided + if is_vision: + try: + base64_img = self.tensor_to_base64(image_input[0]) + messages.append({ + "role": "user", + "content": [ + {"type": "text", "text": prompt_input}, + {"type": "image_url", "image_url": {"url": f"data:image/png;base64,{base64_img}"}} + ] + }) + except Exception as e: + return (f"[ERROR] Failed to encode image: {str(e)}",) + else: + messages.append({"role": "user", "content": prompt_input}) + + payload = { + "model": model_name, + "messages": messages, + "temperature": 0.7, + "max_tokens": 1024, + } + + try: + response = requests.post(lm_api_url, json=payload, timeout=200) + response.raise_for_status() + result = response.json() + content = result.get("choices", [{}])[0].get("message", {}).get("content", "") + return (content.strip(),) + except Exception as e: + return (f"[ERROR] LM Studio API error: {str(e)}",) + +# === Node Registration === +NODE_CLASS_MAPPINGS = { + "PromptGenerator": LMStudioMultimodalPrompt +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "PromptGenerator": "🎛️ Prompt Generator" +} diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..64804f7 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,5 @@ +torch>=2.0 +numpy>=1.24 +opencv-python>=4.7 +Pillow>=9.0 +matplotlib>=3.7