From f1e0d57af8d862d59cc4d43922c58cdd0f01eb22 Mon Sep 17 00:00:00 2001 From: ru4ls <126259521+ru4ls@users.noreply.github.com> Date: Fri, 26 Sep 2025 09:25:30 +0700 Subject: [PATCH] feat: add Image-to-Image generation support with new WanI2IGenerator and API integration, add Wan 2.5 model preview --- README.md | 49 +++++++-- __init__.py | 3 + core/base.py | 2 + generators/i2i.py | 261 ++++++++++++++++++++++++++++++++++++++++++++++ generators/i2v.py | 1 + generators/t2i.py | 8 +- generators/t2v.py | 7 +- 7 files changed, 319 insertions(+), 12 deletions(-) create mode 100644 generators/i2i.py diff --git a/README.md b/README.md index bcd9fa9..c685aaa 100644 --- a/README.md +++ b/README.md @@ -11,10 +11,10 @@ This is a direct integration with Alibaba Cloud's Model Studio service, not a th - **Enterprise-Grade Infrastructure**: Leverages Alibaba Cloud's battle-tested AI platform serving millions of requests daily - **State-of-the-Art Models**: Access to the latest Wan models with continuous updates: - - **Text-to-Image**: wan2.2-t2i-flash (Speed Edition), wan2.2-t2i-plus (Professional Edition) - - **Image-to-Video**: wan2.2-i2v-flash (Speed Edition), wan2.2-i2v-plus (Professional Edition) + - **Text-to-Image**: wan2.5-t2i-preview (Preview Edition), wan2.2-t2i-flash (Speed Edition), wan2.2-t2i-plus (Professional Edition), wanx2.1-t2i-turbo (Turbo Edition), wanx2.1-t2i-plus (Plus Edition), wanx2.0-t2i-turbo (Turbo Edition) + - **Image-to-Video**: wan2.5-i2v-preview (Preview Edition), wan2.2-i2v-flash (Speed Edition), wan2.2-i2v-plus (Professional Edition) - **Image-to-Video Effects**: wan2.1-i2v-plus (Professional Edition) - - **Text-to-Video**: wan2.2-t2v-plus (Professional Edition) + - **Text-to-Video**: wan2.5-t2v-preview (Preview Edition), wan2.2-t2v-plus (Professional Edition), wanx2.1-t2v-turbo (Turbo Edition), wanx2.1-t2v-plus (Plus Edition) - **Image-to-Video (First/Last Frames)**: wan2.1-kf2v-plus (Professional Edition) - **Universal Video Editing (VACE)**: wan2.1-vace-plus (Professional Edition) - Split into 5 specialized nodes for better usability - **Commercial Licensing**: Properly licensed for commercial use through Alibaba Cloud's terms of service @@ -42,10 +42,11 @@ All API endpoints are centrally managed in the `core/base.py` file, making it ea | Node Name | Function | Model | Description | |-----------|----------|-------|-------------| -| Wan Text-to-Image Generator | T2I | wan2.2-t2i-flash, wan2.2-t2i-plus | Generate images from text prompts with multiple resolution options. Returns both image tensor and image URL. | -| Wan Image-to-Video Generator | I2V | wan2.2-i2v-flash, wan2.2-i2v-plus | Create 5-second videos from a single image and text prompt. Returns both video file path and video URL. | +| Wan Text-to-Image Generator | T2I | wan2.5-t2i-preview, wan2.2-t2i-flash, wan2.2-t2i-plus, wanx2.1-t2i-turbo, wanx2.1-t2i-plus, wanx2.0-t2i-turbo | Generate images from text prompts with multiple resolution options. Returns both image tensor and image URL. | +| Wan Image-to-Image Generator | I2I | wan2.5-i2i-preview | Edit images using text prompts and reference images with multiple size options. Returns both image tensor and image URL. | +| Wan Image-to-Video Generator | I2V | wan2.5-i2v-preview, wan2.2-i2v-flash, wan2.2-i2v-plus | Create 5-second videos from a single image and text prompt. Returns both video file path and video URL. | | Wan Image-to-Video Effect Generator | I2V Effect | wan2.1-i2v-plus | Generate videos with predefined effects from a single image. Returns both video file path and video URL. | -| Wan Text-to-Video Generator | T2V | wan2.2-t2v-plus | Generate 5-second videos directly from text prompts. Returns both video file path and video URL. | +| Wan Text-to-Video Generator | T2V | wan2.5-t2v-preview, wan2.2-t2v-plus, wanx2.1-t2v-turbo, wanx2.1-t2v-plus | Generate 5-second videos directly from text prompts. Returns both video file path and video URL. | | Wan Image-to-Video (First/Last Frame) Generator | II2V | wan2.1-kf2v-plus | Create 5-second videos using both first and last frame images. Returns both video file path and video URL. | | Wan VACE - Multi-Image Reference | VACE | wan2.1-vace-plus | Generate videos from multiple reference images. Returns both video file path and video URL. | | Wan VACE - Video Repainting | VACE | wan2.1-vace-plus | Repaint videos while preserving motion. Returns both video file path and video URL. | @@ -110,7 +111,7 @@ If you only use the international region, you only need to set `DASHSCOPE_API_KE ## Node Parameters ### Text-to-Image Generator -- **model**: Select the Wan model to use (wan2.2-t2i-flash or wan2.2-t2i-plus) +- **model**: Select the Wan model to use (wan2.5-t2i-preview, wan2.2-t2i-flash, wan2.2-t2i-plus, wanx2.1-t2i-turbo, wanx2.1-t2i-plus, wanx2.0-t2i-turbo) - **prompt** (required): The text prompt for image generation - **size**: Output image resolution (1024×1024, 1152×896, 896×1152, 1280×720, 720×1280, 1440×512, 512×1440) - **negative_prompt**: Text describing content to avoid in the image @@ -122,10 +123,25 @@ If you only use the international region, you only need to set `DASHSCOPE_API_KE - **image**: Generated image as a tensor (can be connected directly to other ComfyUI nodes) - **image_url**: URL of the generated image on Alibaba Cloud's servers +### Image-to-Image Generator +- **model**: Select the Wan model to use (wan2.5-i2i-preview) +- **image_url_1** (required): Publicly accessible URL to the first input image for editing +- **image_url_2** (optional): Publicly accessible URL to the second reference image (for multi-reference generation) +- **prompt** (required): The text prompt describing the desired changes to the image +- **size**: Output image resolution (1024×1024, 1152×896, 896×1152, 1280×720, 720×1280, 1440×512, 512×1440) +- **negative_prompt**: Text describing content to avoid in the edited image +- **watermark**: Add Wan watermark to output +- **seed**: Random seed for generation (0 for random) +- **num_images**: Number of images to generate (1-4) + +**Return Values:** +- **image**: Generated image as a tensor (can be connected directly to other ComfyUI nodes) +- **image_url**: URL of the generated image on Alibaba Cloud's servers + ### Text-to-Video Generator -- **model**: Select the Wan model to use (wan2.2-t2v-plus) +- **model**: Select the Wan model to use (wan2.5-t2v-preview, wan2.2-t2v-plus, wanx2.1-t2v-turbo, wanx2.1-t2v-plus) - **prompt** (required): The text prompt for video generation -- **resolution**: Output video resolution (480P, 1080P) +- **resolution**: Output video resolution (480P, 720P, 1080P) - **negative_prompt**: Text describing content to avoid in the video - **prompt_extend**: Enable intelligent prompt rewriting for better results - **seed**: Random seed for generation (0 for random) @@ -139,7 +155,7 @@ If you only use the international region, you only need to set `DASHSCOPE_API_KE **Note**: To preview the generated video in ComfyUI, connect the output of this node to a "Load Video (Path)" node from ComfyUI-VideoHelperSuite. ### Image-to-Video Generator -- **model**: Select the Wan model to use (wan2.2-i2v-flash or wan2.2-i2v-plus) +- **model**: Select the Wan model to use (wan2.5-i2v-preview, wan2.2-i2v-flash, wan2.2-i2v-plus) - **image_url**: Publicly accessible URL to the image for the first frame of the video - **prompt** (required): The text prompt describing the video content - **resolution**: Output video resolution (480P, 720P, 1080P) @@ -330,6 +346,19 @@ The API key is loaded from the `DASHSCOPE_API_KEY` environment variable and neve ## Changelog +### v1.3.0 - Image-to-Image Support +- Added new Wan Image-to-Image Generator node supporting wan2.5-i2i-preview model +- Implemented complete i2i functionality with multi-image reference support +- Added i2i API endpoint to core/base.py for proper communication +- Updated documentation with new node and its parameters +- Enhanced Available Nodes table to include Image-to-Image capabilities + +### v1.2.0 - Model Updates +- Added new Wan2.5 Preview models: wan2.5-t2i-preview, wan2.5-i2v-preview, wan2.5-t2v-preview +- Added WanX models: wanx2.1-t2i-turbo, wanx2.1-t2i-plus, wanx2.1-t2v-turbo, wanx2.1-t2v-plus, wanx2.0-t2i-turbo +- Updated model lists across all generator nodes (T2I, I2V, T2V) to include the latest Wan models +- Enhanced documentation with updated model information and capabilities + ### v1.1.0 - Region Selection Feature - Added region selection parameter to all nodes, allowing users to easily switch between international and Mainland China regions - Updated `.env.template` to include separate API key variables for international and Mainland China regions diff --git a/__init__.py b/__init__.py index 1b42f06..32fcf6d 100644 --- a/__init__.py +++ b/__init__.py @@ -8,6 +8,7 @@ from .generators.i2v import WanI2VGenerator from .generators.i2v_effect import WanI2VEffectGenerator from .generators.t2v import WanT2VGenerator from .generators.ii2v import WanII2VGenerator +from .generators.i2i import WanI2IGenerator from .vace.image_reference import WanVACEImageReference from .vace.video_repainting import WanVACEVideoRepainting from .vace.video_edit import WanVACEVideoEdit @@ -20,6 +21,7 @@ NODE_CLASS_MAPPINGS = { "WanI2VEffectGenerator": WanI2VEffectGenerator, "WanT2VGenerator": WanT2VGenerator, "WanII2VGenerator": WanII2VGenerator, + "WanI2IGenerator": WanI2IGenerator, "WanVACEImageReference": WanVACEImageReference, "WanVACEVideoRepainting": WanVACEVideoRepainting, "WanVACEVideoEdit": WanVACEVideoEdit, @@ -33,6 +35,7 @@ NODE_DISPLAY_NAME_MAPPINGS = { "WanI2VEffectGenerator": "Wan Image-to-Video Effect Generator", "WanT2VGenerator": "Wan Text-to-Video Generator", "WanII2VGenerator": "Wan Image-to-Video (First/Last Frame) Generator", + "WanI2IGenerator": "Wan Image-to-Image Generator", "WanVACEImageReference": "Wan VACE - Multi-Image Reference", "WanVACEVideoRepainting": "Wan VACE - Video Repainting", "WanVACEVideoEdit": "Wan VACE - Local Video Editing", diff --git a/core/base.py b/core/base.py index bb6b8ac..93afd75 100644 --- a/core/base.py +++ b/core/base.py @@ -49,12 +49,14 @@ class WanAPIBase: "video_post": "https://dashscope-intl.aliyuncs.com/api/v1/services/aigc/video-generation/video-synthesis", "ii2v_post": "https://dashscope-intl.aliyuncs.com/api/v1/services/aigc/image2video/video-synthesis", "t2i_post": "https://dashscope-intl.aliyuncs.com/api/v1/services/aigc/text2image/image-synthesis", + "i2i_post": "https://dashscope-intl.aliyuncs.com/api/v1/services/aigc/image2image/image-synthesis", "get": "https://dashscope-intl.aliyuncs.com/api/v1/tasks/{task_id}" }, "mainland_china": { "video_post": "https://dashscope.aliyuncs.com/api/v1/services/aigc/video-generation/video-synthesis", "ii2v_post": "https://dashscope.aliyuncs.com/api/v1/services/aigc/image2video/video-synthesis", "t2i_post": "https://dashscope.aliyuncs.com/api/v1/services/aigc/text2image/image-synthesis", + "i2i_post": "https://dashscope.aliyuncs.com/api/v1/services/aigc/image2image/image-synthesis", "get": "https://dashscope.aliyuncs.com/api/v1/tasks/{task_id}" } } diff --git a/generators/i2i.py b/generators/i2i.py new file mode 100644 index 0000000..6daf3d5 --- /dev/null +++ b/generators/i2i.py @@ -0,0 +1,261 @@ +import os +import json +import requests +from PIL import Image +import numpy as np +import torch +import io +import base64 +from dotenv import load_dotenv +import sys +import pathlib +from datetime import datetime + +# Import the base class and COMFYUI_AVAILABLE flag +from ..core.base import WanAPIBase, COMFYUI_AVAILABLE + +# Try to import folder_paths if available +try: + import folder_paths +except ImportError: + pass + +class WanI2IGenerator(WanAPIBase): + """Node for image-to-image generation using Wan model""" + + # Define available Wan i2i models + MODEL_OPTIONS = [ + "wan2.5-i2i-preview", # Preview Edition + ] + + # Define allowed sizes for Wan i2i models + SIZE_OPTIONS = [ + "1024*1024", # 1:1 square (default) + "1152*896", # 9:7 landscape + "896*1152", # 7:9 portrait + "1280*720", # 16:9 landscape + "720*1280", # 9:16 portrait + "1440*512", # Wide landscape + "512*1440", # Tall portrait + "768*768", # 1:1 square + "1440*1440", # 1:1 square + ] + + # Define region options + REGION_OPTIONS = [ + "international", + "mainland_china" + ] + + def __init__(self): + super().__init__() + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "model": (cls.MODEL_OPTIONS, { + "default": "wan2.5-i2i-preview" + }), + "image_url_1": ("STRING", { + "default": "https://example.com/your_image1.png" + }), + "prompt": ("STRING", { + "multiline": True, + "default": "Edit the image with the desired changes" + }), + "region": (cls.REGION_OPTIONS, { + "default": "international" + }) + }, + "optional": { + "image_url_2": ("STRING", { + "default": "" + }), + "negative_prompt": ("STRING", { + "multiline": True, + "default": "" + }), + "size": (cls.SIZE_OPTIONS, { + "default": "1024*1024" + }), + "watermark": ("BOOLEAN", { + "default": False + }), + "seed": ("INT", { + "default": 0, + "min": 0, + "max": 2147483647 + }), + "num_images": ("INT", { + "default": 1, + "min": 1, + "max": 4 + }) + } + } + + RETURN_TYPES = ("IMAGE", "STRING") # Returns image tensor and image URL + RETURN_NAMES = ("image", "image_url") + FUNCTION = "generate" + CATEGORY = "Ru4ls/Wan" + + def generate(self, model, image_url_1, prompt, region, image_url_2="", negative_prompt="", size="1024*1024", + watermark=False, seed=0, num_images=1): + # Check API key based on region + api_key = self.check_api_key(region) + + # Get the appropriate API endpoints based on region + endpoints = self.get_api_endpoints(region) + api_url = endpoints["i2i_post"] + + # Prepare images array - at least one image is required + images = [image_url_1.strip()] + if image_url_2 and image_url_2.strip(): + images.append(image_url_2.strip()) + + # Prepare API payload for image-to-image generation + payload = { + "model": model, + "input": { + "prompt": prompt, + "images": images + }, + "parameters": { + "size": size, + "watermark": watermark, + "n": num_images + } + } + + # Add optional parameters if they have non-default values + if negative_prompt: + payload["input"]["negative_prompt"] = negative_prompt + if seed > 0: + payload["parameters"]["seed"] = seed + + # Set headers according to DashScope documentation + headers = { + "Authorization": f"Bearer {api_key}", + "Content-Type": "application/json", + "X-DashScope-Async": "enable" # Wan requires async processing + } + + try: + # Make API request + print(f"Making API request to {api_url}") + response = requests.post(api_url, headers=headers, json=payload) + print(f"Response status code: {response.status_code}") + if hasattr(response, 'text'): + print(f"Response text: {response.text[:500]}...") # Print first 500 chars + response.raise_for_status() + + # Parse response to get task_id + result = response.json() + print(f"API response received: {json.dumps(result, indent=2)[:200]}...") # Print first 200 chars + + # Check if this is a task creation response + if "output" in result and "task_id" in result["output"]: + task_id = result["output"]["task_id"] + task_status = result["output"]["task_status"] + print(f"Task created with ID: {task_id}, status: {task_status}") + + # Now we need to poll for the result + task_result = self.poll_task_result(task_id, region) + return task_result # Return both image tensor and image URL + else: + raise ValueError(f"Unexpected API response format: {result}") + + except requests.exceptions.RequestException as e: + # More detailed error handling + if hasattr(e, 'response') and e.response is not None: + status_code = e.response.status_code + response_text = e.response.text + print(f"API request failed with status {status_code}: {response_text}") + if status_code == 401: + raise RuntimeError(f"API request failed: 401 Unauthorized. " + f"This usually means your API key is invalid or not properly configured. " + f"Error details: {response_text}") + elif status_code == 403: + raise RuntimeError(f"API request failed: 403 Forbidden. " + f"This usually means your API key is valid but you don't have access to this model. " + f"Error details: {response_text}") + elif status_code == 400: + raise RuntimeError(f"API request failed: 400 Bad Request. " + f"This usually means there's an issue with the request format. " + f"Error details: {response_text}") + else: + raise RuntimeError(f"API request failed: {status_code} {e.response.reason}. Response: {response_text}") + else: + raise RuntimeError(f"API request failed: {str(e)}") + except Exception as e: + raise RuntimeError(f"Failed to process API response: {str(e)}") + + def poll_task_result(self, task_id, region): + """Poll for task result until completion""" + import time + + # Get the appropriate API endpoints based on region + endpoints = self.get_api_endpoints(region) + query_url = endpoints["get"].format(task_id=task_id) + + # Check API key based on region + api_key = self.check_api_key(region) + + headers = { + "Authorization": f"Bearer {api_key}", + "Content-Type": "application/json" + } + + max_attempts = 30 # Maximum polling attempts + attempt = 0 + + while attempt < max_attempts: + try: + print(f"Polling task {task_id}, attempt {attempt + 1}/{max_attempts}") + response = requests.get(query_url, headers=headers) + response.raise_for_status() + + result = response.json() + task_status = result["output"]["task_status"] + print(f"Task status: {task_status}") + + if task_status == "SUCCEEDED": + # Task completed successfully + results = result["output"]["results"] + if len(results) > 0 and "url" in results[0]: + image_url = results[0]["url"] + # Download the generated image + image_response = requests.get(image_url) + image_response.raise_for_status() + + # Convert to tensor + image = Image.open(io.BytesIO(image_response.content)) + image_tensor = torch.from_numpy(np.array(image).astype(np.float32) / 255.0) + image_tensor = image_tensor.unsqueeze(0) # Add batch dimension + + # Return both the image tensor and the image URL + return (image_tensor, image_url) + else: + raise ValueError(f"Unexpected API response format: {result}") + + elif task_status == "FAILED": + # Task failed + error_code = result["output"].get("code", "Unknown") + error_message = result["output"].get("message", "Unknown error") + raise RuntimeError(f"Task failed with code: {error_code}, message: {error_message}") + + elif task_status in ["PENDING", "RUNNING"]: + # Task still in progress, wait and retry + time.sleep(5) # Wait 5 seconds before retrying + attempt += 1 + continue + + else: + raise ValueError(f"Unexpected task status: {task_status}") + + except requests.exceptions.RequestException as e: + raise RuntimeError(f"Failed to query task status: {str(e)}") + + # If we've reached here, we've exceeded max attempts + raise RuntimeError(f"Task did not complete within the expected time ({max_attempts} attempts)") \ No newline at end of file diff --git a/generators/i2v.py b/generators/i2v.py index 9704b2e..1f9882d 100644 --- a/generators/i2v.py +++ b/generators/i2v.py @@ -25,6 +25,7 @@ class WanI2VGenerator(WanAPIBase): # Define available Wan i2v models MODEL_OPTIONS = [ + "wan2.5-i2v-preview", # Preview Edition "wan2.2-i2v-flash", # Speed Edition "wan2.2-i2v-plus" # Professional Edition ] diff --git a/generators/t2i.py b/generators/t2i.py index c656b01..5a5a82a 100644 --- a/generators/t2i.py +++ b/generators/t2i.py @@ -24,8 +24,12 @@ class WanT2IGenerator(WanAPIBase): # Define available Wan models MODEL_OPTIONS = [ + "wan2.5-t2i-preview", # Preview Edition "wan2.2-t2i-flash", # Speed Edition - "wan2.2-t2i-plus" # Professional Edition + "wan2.2-t2i-plus", # Professional Edition + "wanx2.1-t2i-turbo" # Turbo Edition + "wanx2.1-t2i-plus" # Plus Edition + "wanx2.0-t2i-turbo" # Turbo Edition ] # Define allowed sizes for Wan models with descriptive names @@ -38,6 +42,8 @@ class WanT2IGenerator(WanAPIBase): "720*1280", # 9:16 portrait "1440*512", # Wide landscape "512*1440" # Tall portrait + "768*768" # 1:1 square + "1440*1440" # 1:1 square ] # Define region options diff --git a/generators/t2v.py b/generators/t2v.py index 0bd3b56..ee5eac6 100644 --- a/generators/t2v.py +++ b/generators/t2v.py @@ -25,12 +25,17 @@ class WanT2VGenerator(WanAPIBase): # Define available Wan t2v models MODEL_OPTIONS = [ - "wan2.2-t2v-plus" # Professional Edition + "wan2.5-t2v-preview", # Preview Edition + "wan2.2-t2v-plus", # Professional Edition + "wanx2.1-t2v-turbo", # Turbo Edition + "wanx2.1-t2v-plus", # Plus Edition + ] # Define allowed resolutions for Wan t2v models RESOLUTION_OPTIONS = [ "480P", + "720P", "1080P" ]