From 43b3cd807686cf359da7a2d8dcef836fce806849 Mon Sep 17 00:00:00 2001 From: Maxed-Out-99 Date: Fri, 16 Jan 2026 21:41:58 -0800 Subject: [PATCH] Add Qwen image edit nodes and update outputs Introduces QwenImageEditSingleMXD and QwenImageEditTripleMXD nodes with support for outputting both conditioning and latent tensors. Removes the ControlNetPreprocessorSelector node. Updates display names and output schemas for Qwen nodes, and ensures image previews are shown in the UI for MxdImageComparerSave. Adds a 'testing/' directory to .gitignore and integrates test node mappings in __init__.py. --- .gitignore | 3 + __init__.py | 6 ++ maxedoutnodes.py | 176 +++++++++++++++++++++++++++++++--------------- mediacomparers.py | 2 + 4 files changed, 132 insertions(+), 55 deletions(-) diff --git a/.gitignore b/.gitignore index 7e2b140..b74de8b 100644 --- a/.gitignore +++ b/.gitignore @@ -28,6 +28,9 @@ inputs/ runs/ logs/ +# Local testing workspace +testing/ + # ComfyUI local cache & configs ComfyUI/output/ ComfyUI/input/ diff --git a/__init__.py b/__init__.py index 2b05c56..b79bde5 100644 --- a/__init__.py +++ b/__init__.py @@ -10,6 +10,10 @@ from .wan22nodes import ( NODE_CLASS_MAPPINGS as WAN22_NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as WAN22_NODE_DISPLAY_NAME_MAPPINGS, ) +from .testing.testnodes import ( + NODE_CLASS_MAPPINGS as TEST_NODE_CLASS_MAPPINGS, + NODE_DISPLAY_NAME_MAPPINGS as TEST_NODE_DISPLAY_NAME_MAPPINGS, +) WEB_DIRECTORY = "web" @@ -18,11 +22,13 @@ NODE_CLASS_MAPPINGS = {} NODE_CLASS_MAPPINGS.update(MXD_NODE_CLASS_MAPPINGS) NODE_CLASS_MAPPINGS.update(MEDIA_NODE_CLASS_MAPPINGS) NODE_CLASS_MAPPINGS.update(WAN22_NODE_CLASS_MAPPINGS) +NODE_CLASS_MAPPINGS.update(TEST_NODE_CLASS_MAPPINGS) NODE_DISPLAY_NAME_MAPPINGS = {} NODE_DISPLAY_NAME_MAPPINGS.update(MXD_NODE_DISPLAY_NAME_MAPPINGS) NODE_DISPLAY_NAME_MAPPINGS.update(MEDIA_NODE_DISPLAY_NAME_MAPPINGS) NODE_DISPLAY_NAME_MAPPINGS.update(WAN22_NODE_DISPLAY_NAME_MAPPINGS) +NODE_DISPLAY_NAME_MAPPINGS.update(TEST_NODE_DISPLAY_NAME_MAPPINGS) __all__ = [ "NODE_CLASS_MAPPINGS", diff --git a/maxedoutnodes.py b/maxedoutnodes.py index a5e7d93..1a98bd4 100644 --- a/maxedoutnodes.py +++ b/maxedoutnodes.py @@ -101,40 +101,6 @@ class FluxResolutionSelector: def select_resolution(self, resolution) -> tuple: return (resolution,) -######################################################################################################################## -# near your selector -PREPROCESSOR_OPTIONS = [ - "LineArtPreprocessor", - "CannyEdgePreprocessor", - "M-LSDPreprocessor", -] - -try: - # adjust import path to where your AIO node actually lives - from custom_nodes.comfyui_controlnet_aux.__init__ import AIO_Preprocessor - FULL_PREPROCESSOR_OPTIONS = list( - AIO_Preprocessor.INPUT_TYPES()["required"]["preprocessor"][0] - ) -except Exception: - # fallback so the node still loads if AIO isn't available - FULL_PREPROCESSOR_OPTIONS = PREPROCESSOR_OPTIONS - -class ControlNetPreprocessorSelector: - DESCRIPTION = """Select a ControlNet preprocessor from a short list.""" - @classmethod - def INPUT_TYPES(cls): - return { - "required": { - "preprocessor": (PREPROCESSOR_OPTIONS, {"default": "LineArtPreprocessor"}) - } - } - - RETURN_TYPES = (FULL_PREPROCESSOR_OPTIONS,) - RETURN_NAMES = ("preprocessor",) - - def get_preprocessor(self, preprocessor): - return (preprocessor,) - ######################################################################################################################## # Sdxl Empty Latent Image class SdxlEmptyLatentImage: @@ -428,64 +394,164 @@ class PromptWithGuidance(ComfyNodeABC): return (conditioning,) ######################################################################################################################## -# Qwen Image Edit Single MXD class QwenImageEditSingleMXD(io.ComfyNode): @classmethod def define_schema(cls): return io.Schema( node_id="QwenImageEditSingleMXD", - display_name="Qwen Image Edit Prompt Single MXD", + display_name="Qwen Image Edit + Latent MXD", category="MXD/conditioning", - description="Encode a prompt and optional image for Qwen single-image editing.", + description="Encode prompt/image and output a matching empty latent.", inputs=[ io.Clip.Input("clip"), io.String.Input("prompt", multiline=True, dynamic_prompts=True), io.Vae.Input("vae", optional=True), io.Image.Input("image", optional=True), + io.Int.Input("batch_size", default=1, min=1, max=4096), ], outputs=[ io.Conditioning.Output(), + io.Latent.Output(), # New Output ], ) @classmethod - def execute(cls, clip, prompt, vae=None, image=None) -> io.NodeOutput: + def execute(cls, clip, prompt, vae=None, image=None, batch_size=1) -> io.NodeOutput: ref_latents = [] images_vl = [] llama_template = "<|im_start|>system\nDescribe the key features of the input image (color, shape, size, texture, objects, background), then explain how the user's text instruction should alter or modify the image. Generate a new image that meets the user's requirements while maintaining consistency with the original input where appropriate.<|im_end|>\n<|im_start|>user\n{}<|im_end|>\n<|im_start|>assistant\n" image_prompt = "" + # Default fallback size if no image is provided (1024x1024) + final_width, final_height = 1024, 1024 + if image is not None: samples = image.movedim(-1, 1) - total = int(384 * 384) + + # --- VISION SCALING (384px area) --- + total_vl = int(384 * 384) + scale_vl = math.sqrt(total_vl / (samples.shape[3] * samples.shape[2])) + width_vl = round(samples.shape[3] * scale_vl) + height_vl = round(samples.shape[2] * scale_vl) - scale_by = math.sqrt(total / (samples.shape[3] * samples.shape[2])) - width = round(samples.shape[3] * scale_by) - height = round(samples.shape[2] * scale_by) + s_vl = comfy.utils.common_upscale(samples, width_vl, height_vl, "area", "disabled") + images_vl.append(s_vl.movedim(1, -1)) + + # --- LATENT/VAE SCALING (1024px area) --- + total_lat = int(1024 * 1024) + scale_lat = math.sqrt(total_lat / (samples.shape[3] * samples.shape[2])) + # Calculate final dimensions to be multiples of 8 + final_width = round(samples.shape[3] * scale_lat / 8.0) * 8 + final_height = round(samples.shape[2] * scale_lat / 8.0) * 8 - s = comfy.utils.common_upscale(samples, width, height, "area", "disabled") - images_vl.append(s.movedim(1, -1)) if vae is not None: - total = int(1024 * 1024) - scale_by = math.sqrt(total / (samples.shape[3] * samples.shape[2])) - width = round(samples.shape[3] * scale_by / 8.0) * 8 - height = round(samples.shape[2] * scale_by / 8.0) * 8 - - s = comfy.utils.common_upscale(samples, width, height, "area", "disabled") - ref_latents.append(vae.encode(s.movedim(1, -1)[:, :, :, :3])) + s_lat = comfy.utils.common_upscale(samples, final_width, final_height, "area", "disabled") + ref_latents.append(vae.encode(s_lat.movedim(1, -1)[:, :, :, :3])) image_prompt += "Picture 1: <|vision_start|><|image_pad|><|vision_end|>" + # 1. Generate the Empty Latent (SD3 Style: 16 channels, 1/8th resolution) + # This replaces the need for the separate EmptySD3LatentImage node + latent_tensor = torch.zeros( + [batch_size, 16, final_height // 8, final_width // 8], + device=comfy.model_management.intermediate_device() + ) + latent_output = {"samples": latent_tensor} + + # 2. Process Conditioning tokens = clip.tokenize(image_prompt + prompt, images=images_vl, llama_template=llama_template) conditioning = clip.encode_from_tokens_scheduled(tokens) + if len(ref_latents) > 0: conditioning = node_helpers.conditioning_set_values( conditioning, {"reference_latents": ref_latents}, append=True, ) - return io.NodeOutput(conditioning) + + return io.NodeOutput(conditioning, latent_output) + + +######################################################################################################################## +class QwenImageEditTripleMXD(io.ComfyNode): + @classmethod + def define_schema(cls): + return io.Schema( + node_id="QwenImageEditTripleMXD", + display_name="Qwen Image Edit Prompt MXD (Triple)", + category="advanced/conditioning", + inputs=[ + io.Clip.Input("clip"), + io.String.Input("prompt", multiline=True, dynamic_prompts=True), + io.Vae.Input("vae", optional=True), + io.Image.Input("image1", optional=True), + io.Image.Input("image2", optional=True), + io.Image.Input("image3", optional=True), + io.Int.Input("batch_size", default=1, min=1, max=4096), + ], + outputs=[ + io.Conditioning.Output(), + io.Latent.Output(), + ], + ) + + @classmethod + def execute(cls, clip, prompt, vae=None, image1=None, image2=None, image3=None, batch_size=1) -> io.NodeOutput: + ref_latents = [] + images = [image1, image2, image3] + images_vl = [] + llama_template = "<|im_start|>system\nDescribe the key features of the input image (color, shape, size, texture, objects, background), then explain how the user's text instruction should alter or modify the image. Generate a new image that meets the user's requirements while maintaining consistency with the original input where appropriate.<|im_end|>\n<|im_start|>user\n{}<|im_end|>\n<|im_start|>assistant\n" + image_prompt = "" + + # Default fallback + latent_width = 1024 + latent_height = 1024 + + for i, image in enumerate(images): + if image is not None: + samples = image.movedim(-1, 1) + + # 1. VL Model Scaling (LLM Vision) + total_vl = int(384 * 384) + scale_by_vl = math.sqrt(total_vl / (samples.shape[3] * samples.shape[2])) + width_vl = round(samples.shape[3] * scale_by_vl) + height_vl = round(samples.shape[2] * scale_by_vl) + s_vl = comfy.utils.common_upscale(samples, width_vl, height_vl, "area", "disabled") + images_vl.append(s_vl.movedim(1, -1)) + + # 2. VAE Scaling (Synchronized to 16-step for SD3 compatibility) + if vae is not None: + total_ref = int(1024 * 1024) + scale_by_ref = math.sqrt(total_ref / (samples.shape[3] * samples.shape[2])) + + # Pixels as multiple of 16 ensures Latent (Pixels/8) is always even + width_ref = round(samples.shape[3] * scale_by_ref / 16.0) * 16 + height_ref = round(samples.shape[2] * scale_by_ref / 16.0) * 16 + + if i == 0: + latent_width = width_ref + latent_height = height_ref + + s_ref = comfy.utils.common_upscale(samples, width_ref, height_ref, "area", "disabled") + ref_latents.append(vae.encode(s_ref.movedim(1, -1)[:, :, :, :3])) + + image_prompt += "Picture {}: <|vision_start|><|image_pad|><|vision_end|>".format(i + 1) + + # Process tokens and conditioning + tokens = clip.tokenize(image_prompt + prompt, images=images_vl, llama_template=llama_template) + conditioning = clip.encode_from_tokens_scheduled(tokens) + + if len(ref_latents) > 0: + conditioning = node_helpers.conditioning_set_values(conditioning, {"reference_latents": ref_latents}, append=True) + + # Create Output Latent + latent = torch.zeros([batch_size, 16, latent_height // 8, latent_width // 8], device=comfy.model_management.intermediate_device()) + + # FIXED: Return outputs positionally to match the schema defined above + # Output 1: Conditioning, Output 2: Latent Dictionary + return io.NodeOutput(conditioning, {"samples": latent}) + ######################################################################################################################## class FluxResolutionMatcher: DESCRIPTION = """Match the closest Flux resolution and orientation for the input image.""" @@ -1136,11 +1202,11 @@ NODE_CLASS_MAPPINGS = { "Flux Empty Latent Image": FluxEmptyLatentImage, "Sdxl Empty Latent Image": SdxlEmptyLatentImage, "Flux Resolution Selector": FluxResolutionSelector, - "ControlNet Preprocessor Selector": ControlNetPreprocessorSelector, "Image Scale To Total Pixels (SDXL Safe)": SDXLImageScaleToTotalPixelsSafe, "Flux Image Scale To Total Pixels (Flux Safe)": FluxImageScaleToTotalPixelsSafe, "Prompt With Guidance (Flux)": PromptWithGuidance, "QwenImageEditSingleMXD": QwenImageEditSingleMXD, + "QwenImageEditTripleMXD": QwenImageEditTripleMXD, "FluxResolutionMatcher": FluxResolutionMatcher, "SDXLResolutionMatcher": SDXLResolutionMatcher, "LatentHalfMasks": LatentHalfMasks, @@ -1156,11 +1222,11 @@ NODE_DISPLAY_NAME_MAPPINGS = { "Flux Empty Latent Image": "Flux Empty Latent Image MXD", "Sdxl Empty Latent Image": "SDXL Empty Latent Image MXD", "Flux Resolution Selector": "Flux Resolution Selector MXD", - "ControlNet Preprocessor Selector": "ControlNet Preprocessor Selector MXD", "Image Scale To Total Pixels (SDXL Safe)": "Scale SDXL Image MXD", "Flux Image Scale To Total Pixels (Flux Safe)": "Scale Flux Image MXD", "Prompt With Guidance (Flux)": "Prompt with Flux Guidance MXD", - "QwenImageEditSingleMXD": "Qwen Image Edit Prompt Single MXD", + "QwenImageEditSingleMXD": "Qwen Image Edit + Latent MXD", + "QwenImageEditTripleMXD": "Qwen Image Edit Prompt MXD (Triple)", "FluxResolutionMatcher": "Flux Resolution Matcher MXD", "SDXLResolutionMatcher": "SDXL Resolution Matcher MXD", "LatentHalfMasks": "Latent to L/R Masks MXD", diff --git a/mediacomparers.py b/mediacomparers.py index df3a885..bd767d4 100644 --- a/mediacomparers.py +++ b/mediacomparers.py @@ -113,6 +113,8 @@ class MxdImageComparerSave(PreviewImage): ) if preview: + # Also publish standard ui.images so the sidebar shows previews. + result_ui["images"] = result_ui["a_images"] + result_ui["b_images"] return {"ui": result_ui} return {}