Add Qwen image edit nodes and update outputs

Introduces QwenImageEditSingleMXD and QwenImageEditTripleMXD nodes with support for outputting both conditioning and latent tensors. Removes the ControlNetPreprocessorSelector node. Updates display names and output schemas for Qwen nodes, and ensures image previews are shown in the UI for MxdImageComparerSave. Adds a 'testing/' directory to .gitignore and integrates test node mappings in __init__.py.
This commit is contained in:
Maxed-Out-99
2026-01-16 21:41:58 -08:00
parent 8672eed53b
commit 43b3cd8076
4 changed files with 132 additions and 55 deletions
+3
View File
@@ -28,6 +28,9 @@ inputs/
runs/
logs/
# Local testing workspace
testing/
# ComfyUI local cache & configs
ComfyUI/output/
ComfyUI/input/
+6
View File
@@ -10,6 +10,10 @@ from .wan22nodes import (
NODE_CLASS_MAPPINGS as WAN22_NODE_CLASS_MAPPINGS,
NODE_DISPLAY_NAME_MAPPINGS as WAN22_NODE_DISPLAY_NAME_MAPPINGS,
)
from .testing.testnodes import (
NODE_CLASS_MAPPINGS as TEST_NODE_CLASS_MAPPINGS,
NODE_DISPLAY_NAME_MAPPINGS as TEST_NODE_DISPLAY_NAME_MAPPINGS,
)
WEB_DIRECTORY = "web"
@@ -18,11 +22,13 @@ NODE_CLASS_MAPPINGS = {}
NODE_CLASS_MAPPINGS.update(MXD_NODE_CLASS_MAPPINGS)
NODE_CLASS_MAPPINGS.update(MEDIA_NODE_CLASS_MAPPINGS)
NODE_CLASS_MAPPINGS.update(WAN22_NODE_CLASS_MAPPINGS)
NODE_CLASS_MAPPINGS.update(TEST_NODE_CLASS_MAPPINGS)
NODE_DISPLAY_NAME_MAPPINGS = {}
NODE_DISPLAY_NAME_MAPPINGS.update(MXD_NODE_DISPLAY_NAME_MAPPINGS)
NODE_DISPLAY_NAME_MAPPINGS.update(MEDIA_NODE_DISPLAY_NAME_MAPPINGS)
NODE_DISPLAY_NAME_MAPPINGS.update(WAN22_NODE_DISPLAY_NAME_MAPPINGS)
NODE_DISPLAY_NAME_MAPPINGS.update(TEST_NODE_DISPLAY_NAME_MAPPINGS)
__all__ = [
"NODE_CLASS_MAPPINGS",
+121 -55
View File
@@ -101,40 +101,6 @@ class FluxResolutionSelector:
def select_resolution(self, resolution) -> tuple:
return (resolution,)
########################################################################################################################
# near your selector
PREPROCESSOR_OPTIONS = [
"LineArtPreprocessor",
"CannyEdgePreprocessor",
"M-LSDPreprocessor",
]
try:
# adjust import path to where your AIO node actually lives
from custom_nodes.comfyui_controlnet_aux.__init__ import AIO_Preprocessor
FULL_PREPROCESSOR_OPTIONS = list(
AIO_Preprocessor.INPUT_TYPES()["required"]["preprocessor"][0]
)
except Exception:
# fallback so the node still loads if AIO isn't available
FULL_PREPROCESSOR_OPTIONS = PREPROCESSOR_OPTIONS
class ControlNetPreprocessorSelector:
DESCRIPTION = """Select a ControlNet preprocessor from a short list."""
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
"preprocessor": (PREPROCESSOR_OPTIONS, {"default": "LineArtPreprocessor"})
}
}
RETURN_TYPES = (FULL_PREPROCESSOR_OPTIONS,)
RETURN_NAMES = ("preprocessor",)
def get_preprocessor(self, preprocessor):
return (preprocessor,)
########################################################################################################################
# Sdxl Empty Latent Image
class SdxlEmptyLatentImage:
@@ -428,64 +394,164 @@ class PromptWithGuidance(ComfyNodeABC):
return (conditioning,)
########################################################################################################################
# Qwen Image Edit Single MXD
class QwenImageEditSingleMXD(io.ComfyNode):
@classmethod
def define_schema(cls):
return io.Schema(
node_id="QwenImageEditSingleMXD",
display_name="Qwen Image Edit Prompt Single MXD",
display_name="Qwen Image Edit + Latent MXD",
category="MXD/conditioning",
description="Encode a prompt and optional image for Qwen single-image editing.",
description="Encode prompt/image and output a matching empty latent.",
inputs=[
io.Clip.Input("clip"),
io.String.Input("prompt", multiline=True, dynamic_prompts=True),
io.Vae.Input("vae", optional=True),
io.Image.Input("image", optional=True),
io.Int.Input("batch_size", default=1, min=1, max=4096),
],
outputs=[
io.Conditioning.Output(),
io.Latent.Output(), # New Output
],
)
@classmethod
def execute(cls, clip, prompt, vae=None, image=None) -> io.NodeOutput:
def execute(cls, clip, prompt, vae=None, image=None, batch_size=1) -> io.NodeOutput:
ref_latents = []
images_vl = []
llama_template = "<|im_start|>system\nDescribe the key features of the input image (color, shape, size, texture, objects, background), then explain how the user's text instruction should alter or modify the image. Generate a new image that meets the user's requirements while maintaining consistency with the original input where appropriate.<|im_end|>\n<|im_start|>user\n{}<|im_end|>\n<|im_start|>assistant\n"
image_prompt = ""
# Default fallback size if no image is provided (1024x1024)
final_width, final_height = 1024, 1024
if image is not None:
samples = image.movedim(-1, 1)
total = int(384 * 384)
# --- VISION SCALING (384px area) ---
total_vl = int(384 * 384)
scale_vl = math.sqrt(total_vl / (samples.shape[3] * samples.shape[2]))
width_vl = round(samples.shape[3] * scale_vl)
height_vl = round(samples.shape[2] * scale_vl)
scale_by = math.sqrt(total / (samples.shape[3] * samples.shape[2]))
width = round(samples.shape[3] * scale_by)
height = round(samples.shape[2] * scale_by)
s_vl = comfy.utils.common_upscale(samples, width_vl, height_vl, "area", "disabled")
images_vl.append(s_vl.movedim(1, -1))
# --- LATENT/VAE SCALING (1024px area) ---
total_lat = int(1024 * 1024)
scale_lat = math.sqrt(total_lat / (samples.shape[3] * samples.shape[2]))
# Calculate final dimensions to be multiples of 8
final_width = round(samples.shape[3] * scale_lat / 8.0) * 8
final_height = round(samples.shape[2] * scale_lat / 8.0) * 8
s = comfy.utils.common_upscale(samples, width, height, "area", "disabled")
images_vl.append(s.movedim(1, -1))
if vae is not None:
total = int(1024 * 1024)
scale_by = math.sqrt(total / (samples.shape[3] * samples.shape[2]))
width = round(samples.shape[3] * scale_by / 8.0) * 8
height = round(samples.shape[2] * scale_by / 8.0) * 8
s = comfy.utils.common_upscale(samples, width, height, "area", "disabled")
ref_latents.append(vae.encode(s.movedim(1, -1)[:, :, :, :3]))
s_lat = comfy.utils.common_upscale(samples, final_width, final_height, "area", "disabled")
ref_latents.append(vae.encode(s_lat.movedim(1, -1)[:, :, :, :3]))
image_prompt += "Picture 1: <|vision_start|><|image_pad|><|vision_end|>"
# 1. Generate the Empty Latent (SD3 Style: 16 channels, 1/8th resolution)
# This replaces the need for the separate EmptySD3LatentImage node
latent_tensor = torch.zeros(
[batch_size, 16, final_height // 8, final_width // 8],
device=comfy.model_management.intermediate_device()
)
latent_output = {"samples": latent_tensor}
# 2. Process Conditioning
tokens = clip.tokenize(image_prompt + prompt, images=images_vl, llama_template=llama_template)
conditioning = clip.encode_from_tokens_scheduled(tokens)
if len(ref_latents) > 0:
conditioning = node_helpers.conditioning_set_values(
conditioning,
{"reference_latents": ref_latents},
append=True,
)
return io.NodeOutput(conditioning)
return io.NodeOutput(conditioning, latent_output)
########################################################################################################################
class QwenImageEditTripleMXD(io.ComfyNode):
@classmethod
def define_schema(cls):
return io.Schema(
node_id="QwenImageEditTripleMXD",
display_name="Qwen Image Edit Prompt MXD (Triple)",
category="advanced/conditioning",
inputs=[
io.Clip.Input("clip"),
io.String.Input("prompt", multiline=True, dynamic_prompts=True),
io.Vae.Input("vae", optional=True),
io.Image.Input("image1", optional=True),
io.Image.Input("image2", optional=True),
io.Image.Input("image3", optional=True),
io.Int.Input("batch_size", default=1, min=1, max=4096),
],
outputs=[
io.Conditioning.Output(),
io.Latent.Output(),
],
)
@classmethod
def execute(cls, clip, prompt, vae=None, image1=None, image2=None, image3=None, batch_size=1) -> io.NodeOutput:
ref_latents = []
images = [image1, image2, image3]
images_vl = []
llama_template = "<|im_start|>system\nDescribe the key features of the input image (color, shape, size, texture, objects, background), then explain how the user's text instruction should alter or modify the image. Generate a new image that meets the user's requirements while maintaining consistency with the original input where appropriate.<|im_end|>\n<|im_start|>user\n{}<|im_end|>\n<|im_start|>assistant\n"
image_prompt = ""
# Default fallback
latent_width = 1024
latent_height = 1024
for i, image in enumerate(images):
if image is not None:
samples = image.movedim(-1, 1)
# 1. VL Model Scaling (LLM Vision)
total_vl = int(384 * 384)
scale_by_vl = math.sqrt(total_vl / (samples.shape[3] * samples.shape[2]))
width_vl = round(samples.shape[3] * scale_by_vl)
height_vl = round(samples.shape[2] * scale_by_vl)
s_vl = comfy.utils.common_upscale(samples, width_vl, height_vl, "area", "disabled")
images_vl.append(s_vl.movedim(1, -1))
# 2. VAE Scaling (Synchronized to 16-step for SD3 compatibility)
if vae is not None:
total_ref = int(1024 * 1024)
scale_by_ref = math.sqrt(total_ref / (samples.shape[3] * samples.shape[2]))
# Pixels as multiple of 16 ensures Latent (Pixels/8) is always even
width_ref = round(samples.shape[3] * scale_by_ref / 16.0) * 16
height_ref = round(samples.shape[2] * scale_by_ref / 16.0) * 16
if i == 0:
latent_width = width_ref
latent_height = height_ref
s_ref = comfy.utils.common_upscale(samples, width_ref, height_ref, "area", "disabled")
ref_latents.append(vae.encode(s_ref.movedim(1, -1)[:, :, :, :3]))
image_prompt += "Picture {}: <|vision_start|><|image_pad|><|vision_end|>".format(i + 1)
# Process tokens and conditioning
tokens = clip.tokenize(image_prompt + prompt, images=images_vl, llama_template=llama_template)
conditioning = clip.encode_from_tokens_scheduled(tokens)
if len(ref_latents) > 0:
conditioning = node_helpers.conditioning_set_values(conditioning, {"reference_latents": ref_latents}, append=True)
# Create Output Latent
latent = torch.zeros([batch_size, 16, latent_height // 8, latent_width // 8], device=comfy.model_management.intermediate_device())
# FIXED: Return outputs positionally to match the schema defined above
# Output 1: Conditioning, Output 2: Latent Dictionary
return io.NodeOutput(conditioning, {"samples": latent})
########################################################################################################################
class FluxResolutionMatcher:
DESCRIPTION = """Match the closest Flux resolution and orientation for the input image."""
@@ -1136,11 +1202,11 @@ NODE_CLASS_MAPPINGS = {
"Flux Empty Latent Image": FluxEmptyLatentImage,
"Sdxl Empty Latent Image": SdxlEmptyLatentImage,
"Flux Resolution Selector": FluxResolutionSelector,
"ControlNet Preprocessor Selector": ControlNetPreprocessorSelector,
"Image Scale To Total Pixels (SDXL Safe)": SDXLImageScaleToTotalPixelsSafe,
"Flux Image Scale To Total Pixels (Flux Safe)": FluxImageScaleToTotalPixelsSafe,
"Prompt With Guidance (Flux)": PromptWithGuidance,
"QwenImageEditSingleMXD": QwenImageEditSingleMXD,
"QwenImageEditTripleMXD": QwenImageEditTripleMXD,
"FluxResolutionMatcher": FluxResolutionMatcher,
"SDXLResolutionMatcher": SDXLResolutionMatcher,
"LatentHalfMasks": LatentHalfMasks,
@@ -1156,11 +1222,11 @@ NODE_DISPLAY_NAME_MAPPINGS = {
"Flux Empty Latent Image": "Flux Empty Latent Image MXD",
"Sdxl Empty Latent Image": "SDXL Empty Latent Image MXD",
"Flux Resolution Selector": "Flux Resolution Selector MXD",
"ControlNet Preprocessor Selector": "ControlNet Preprocessor Selector MXD",
"Image Scale To Total Pixels (SDXL Safe)": "Scale SDXL Image MXD",
"Flux Image Scale To Total Pixels (Flux Safe)": "Scale Flux Image MXD",
"Prompt With Guidance (Flux)": "Prompt with Flux Guidance MXD",
"QwenImageEditSingleMXD": "Qwen Image Edit Prompt Single MXD",
"QwenImageEditSingleMXD": "Qwen Image Edit + Latent MXD",
"QwenImageEditTripleMXD": "Qwen Image Edit Prompt MXD (Triple)",
"FluxResolutionMatcher": "Flux Resolution Matcher MXD",
"SDXLResolutionMatcher": "SDXL Resolution Matcher MXD",
"LatentHalfMasks": "Latent to L/R Masks MXD",
+2
View File
@@ -113,6 +113,8 @@ class MxdImageComparerSave(PreviewImage):
)
if preview:
# Also publish standard ui.images so the sidebar shows previews.
result_ui["images"] = result_ui["a_images"] + result_ui["b_images"]
return {"ui": result_ui}
return {}