diff --git a/README.md b/README.md index e0ec9b9..058f2b9 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # ComfyUI-UniversalToolkit -[![Version](https://img.shields.io/badge/version-1.3.2-blue.svg)](https://github.com/whmc76/ComfyUI-UniversalToolkit) +[![Version](https://img.shields.io/badge/version-1.4.8-blue.svg)](https://github.com/whmc76/ComfyUI-UniversalToolkit) [![License](https://img.shields.io/badge/license-MIT-green.svg)](LICENSE) [![ComfyUI](https://img.shields.io/badge/ComfyUI-v3+-orange.svg)](https://github.com/comfyanonymous/ComfyUI) @@ -52,9 +52,10 @@ tqdm - **ImageConcatenate_UTK**:水平或垂直拼接两张图像 - **ImageConcatenateMulti_UTK**:智能拼接多张图像,支持2-4图自动布局 -#### 图像变换与调整 -- **ImageScaleByAspectRatio_UTK**:按指定宽高比缩放图像 -- **ImageMaskScaleAs_UTK**:按参考图像尺寸缩放图像 +- #### 图像变换与调整 +- **ResizeImageVerKJ_UTK**:KJ v2 风格的高兼容缩放,支持 stretch/resize/pad/pad_edge/pad_edge_pixel/crop/pillarbox_blur/total_pixels 与 `crop_position` +- **ImageScaleByAspectRatio_UTK**:按指定宽高比缩放图像(已支持与 KJ v2 一致的 fit 模式与 `crop_position`,背景色为预设清单) +- **ImageMaskScaleAs_UTK**:按参考图像尺寸缩放图像(已支持与 KJ v2 一致的 fit 模式与 `crop_position`,pad_color 为预设清单) - **ImageScaleRestore_UTK**:将图像恢复到原始尺寸 - **ImageRemoveAlpha_UTK**:移除图像的Alpha通道 - **ImageCombineAlpha_UTK**:合并Alpha通道到图像 @@ -81,6 +82,7 @@ tqdm - **MaskAnd_UTK**:掩码与运算 - **MaskSub_UTK**:掩码减法运算 - **MaskAdd_UTK**:掩码加法运算 +- **BlockifyMask_UTK**:将掩码按 block_size 马赛克化(支持 cpu/cuda;可选二值化) ### 🛠️ 工具节点 @@ -91,6 +93,7 @@ tqdm #### 数学与逻辑 - **MathExpression_UTK**:数学表达式计算,支持复杂公式和函数 +- **BestContextWindow_UTK**:最佳滑动窗口帧数计算(满足 4n+1,最小化补帧;输出 best_window/padding/padded_total/segments) #### 系统工具 - **PurgeVRAM_UTK**:显存清理,支持选择性清理缓存和模型 @@ -176,7 +179,22 @@ AudioCropProcess_UTK ## 📋 版本历史 -### v1.3.2 (最新) +### v1.4.7 (最新) +- 修复 resize 与 pad 方法表现相同的问题: + - ResizeImageVerKJ (UTK):resize 模式只等比缩放不填充,pad 模式填充到目标尺寸 + - ImageMaskScaleAs (UTK):resize 返回实际缩放尺寸,pad 填充到目标尺寸并正确输出尺寸 + - ImageScaleByAspectRatio (UTK):resize 返回实际缩放尺寸,pad 填充到目标尺寸并正确输出尺寸 + - resize:等比缩放,输出尺寸 = 缩放后尺寸(可能小于目标尺寸) + - pad:等比缩放 + 背景填充,输出尺寸 = 目标尺寸(固定尺寸) + +### v1.4.6 +- 新增 `Resize Image ver KJ (UTK)`,完整对齐 KJ v2 调整模式,支持 `crop_position` 与 mask 同步缩放;pad_edge/pad_edge_pixel 行为与 KJ 对齐 +- 升级 `Image Mask Scale As (UTK)` 与 `Image Scale By Aspect Ratio (UTK)`:支持同样的 fit 模式、`crop_position`,并将背景色改为预设清单 +- 新增 `Blockify Mask (UTK)`:掩码块化,支持二值化 +- 新增 `Best Context Window (UTK)`:计算满足 4n+1 的最佳窗口,最小化补帧 +- 统一分类命名:`UniversalToolkit/Tools` + +### v1.3.2 - 新增电商应用类,重新组织预设分类结构 - 创建专门的电商应用类,包含6个专业电商功能: - Ecommerce-Professional Product Photography (专业产品图) diff --git a/__init__.py b/__init__.py index a72fc01..fa1ddef 100644 --- a/__init__.py +++ b/__init__.py @@ -8,13 +8,48 @@ A comprehensive toolkit for ComfyUI that provides various utility nodes for imag :license: MIT, see LICENSE for more details. """ -__version__ = "1.4.4" +__version__ = "1.4.8" __author__ = "CyberDickLang" __email__ = "286878701@qq.com" __url__ = "https://github.com/whmc76" # 更新日志 CHANGELOG = { + "1.4.8": [ + "版本更新和代码优化:", + "- 更新插件版本号为 1.4.8", + "- 代码优化和稳定性改进", + ], + "1.4.7": [ + "修复 resize 与 pad 方法表现相同的问题:", + "- ResizeImageVerKJ (UTK):resize 模式只等比缩放不填充,pad 模式填充到目标尺寸", + "- ImageMaskScaleAs (UTK):resize 返回实际缩放尺寸,pad 填充到目标尺寸并正确输出尺寸", + "- ImageScaleByAspectRatio (UTK):resize 返回实际缩放尺寸,pad 填充到目标尺寸并正确输出尺寸", + "- resize:等比缩放,输出尺寸 = 缩放后尺寸(可能小于目标尺寸)", + "- pad:等比缩放 + 背景填充,输出尺寸 = 目标尺寸(固定尺寸)", + ], + "1.4.6": [ + "新增 Resize Image ver KJ (UTK):", + "- 复刻 KJ v2 的调整模式:stretch/resize/pad/pad_edge/pad_edge_pixel/crop/pillarbox_blur/total_pixels", + "- 支持 mask 同步缩放与对齐,pad_edge/pad_edge_pixel 行为与 KJ 对齐", + "升级 Image Mask Scale As (UTK):", + "- fit 与 KJ v2 对齐,新增 crop_position,支持预设 pad_color", + "升级 Image Scale By Aspect Ratio (UTK):", + "- fit 与 KJ v2 对齐,新增 crop_position,background_color 改为预设清单", + "修正 pad_edge 与 pad_edge_pixel 的边缘与角点处理逻辑,匹配 KJ 视觉表现", + ], + "1.4.5": [ + "新增Best Context Window (UTK)节点:", + "- 计算满足4n+1且位于[min,max]区间的最佳窗口,以最小化补帧", + "- 输出best_window、padding、padded_total、segments", + "- 分类:UniversalToolkit/Tools", + "新增Blockify Mask (UTK)节点:", + "- 将掩码按block_size块化,支持cpu/cuda", + "- 可选二值化binarize与threshold", + "- 分类:UniversalToolkit/Mask", + "统一分类命名:将UniversalToolkit/tools合并为UniversalToolkit/Tools", + "修复:Get Image or Mask Range From Batch (UTK) 分类名不一致问题", + ], "1.4.4": [ "新增Get Image or Mask Range From Batch (UTK)节点:", "- 支持从图像批次或遮罩批次中提取指定范围的元素", @@ -742,9 +777,27 @@ try: NODE_CLASS_MAPPINGS as GET_IMAGE_RANGE_MAPPINGS from .nodes.tools.get_image_range_from_batch import \ NODE_DISPLAY_NAME_MAPPINGS as GET_IMAGE_RANGE_DISPLAY + from .nodes.tools.optimal_context_window_node import \ + NODE_CLASS_MAPPINGS as BEST_CONTEXT_WINDOW_MAPPINGS + from .nodes.tools.optimal_context_window_node import \ + NODE_DISPLAY_NAME_MAPPINGS as BEST_CONTEXT_WINDOW_DISPLAY + from .nodes.mask.blockify_mask import \ + NODE_CLASS_MAPPINGS as BLOCKIFY_MASK_MAPPINGS + from .nodes.mask.blockify_mask import \ + NODE_DISPLAY_NAME_MAPPINGS as BLOCKIFY_MASK_DISPLAY + from .nodes.image.resize_image_ver_kj import \ + NODE_CLASS_MAPPINGS as RESIZE_VER_KJ_MAPPINGS + from .nodes.image.resize_image_ver_kj import \ + NODE_DISPLAY_NAME_MAPPINGS as RESIZE_VER_KJ_DISPLAY except ImportError: GET_IMAGE_RANGE_MAPPINGS = {} GET_IMAGE_RANGE_DISPLAY = {} + BEST_CONTEXT_WINDOW_MAPPINGS = {} + BEST_CONTEXT_WINDOW_DISPLAY = {} + BLOCKIFY_MASK_MAPPINGS = {} + BLOCKIFY_MASK_DISPLAY = {} + RESIZE_VER_KJ_MAPPINGS = {} + RESIZE_VER_KJ_DISPLAY = {} # 合并所有节点映射 @@ -786,6 +839,9 @@ NODE_CLASS_MAPPINGS.update(COLOR_TO_MASK_MAPPINGS) NODE_CLASS_MAPPINGS.update(LAZY_SWITCH_MAPPINGS) NODE_CLASS_MAPPINGS.update(TEXT_TRANSLATOR_API_MAPPINGS) NODE_CLASS_MAPPINGS.update(GET_IMAGE_RANGE_MAPPINGS) +NODE_CLASS_MAPPINGS.update(BEST_CONTEXT_WINDOW_MAPPINGS) +NODE_CLASS_MAPPINGS.update(BLOCKIFY_MASK_MAPPINGS) +NODE_CLASS_MAPPINGS.update(RESIZE_VER_KJ_MAPPINGS) # 合并显示名称映射 NODE_DISPLAY_NAME_MAPPINGS = {} @@ -826,6 +882,9 @@ NODE_DISPLAY_NAME_MAPPINGS.update(COLOR_TO_MASK_DISPLAY) NODE_DISPLAY_NAME_MAPPINGS.update(LAZY_SWITCH_DISPLAY) NODE_DISPLAY_NAME_MAPPINGS.update(TEXT_TRANSLATOR_API_DISPLAY) NODE_DISPLAY_NAME_MAPPINGS.update(GET_IMAGE_RANGE_DISPLAY) +NODE_DISPLAY_NAME_MAPPINGS.update(BEST_CONTEXT_WINDOW_DISPLAY) +NODE_DISPLAY_NAME_MAPPINGS.update(BLOCKIFY_MASK_DISPLAY) +NODE_DISPLAY_NAME_MAPPINGS.update(RESIZE_VER_KJ_DISPLAY) NODE_CATEGORIES = { "UniversalToolkit": [ @@ -867,6 +926,9 @@ NODE_CATEGORIES = { "APIImageGenerator_UTK", "TextTranslatorAPI_UTK", "GetImageRangeFromBatch_UTK", + "BestContextWindow_UTK", + "BlockifyMask_UTK", + "ResizeImageVerKJ_UTK", ] } diff --git a/ize 与 pad 方法表现相同的问题 b/ize 与 pad 方法表现相同的问题 new file mode 100644 index 0000000..7797318 --- /dev/null +++ b/ize 与 pad 方法表现相同的问题 @@ -0,0 +1,338 @@ +diff --git a/README.md b/README.md +index e0ec9b9..2a0873a 100644 +--- a/README.md ++++ b/README.md +@@ -1,6 +1,6 @@ + # ComfyUI-UniversalToolkit +  +-[![Version](https://img.shields.io/badge/version-1.3.2-blue.svg)](https://github.com/whmc76/ComfyUI-UniversalToolkit) ++[![Version](https://img.shields.io/badge/version-1.4.7-blue.svg)](https://github.com/whmc76/ComfyUI-UniversalToolkit) + [![License](https://img.shields.io/badge/license-MIT-green.svg)](LICENSE) + [![ComfyUI](https://img.shields.io/badge/ComfyUI-v3+-orange.svg)](https://github.com/comfyanonymous/ComfyUI) +  +@@ -52,9 +52,10 @@ tqdm + - **ImageConcatenate_UTK**:水平或垂直拼接两张图像 + - **ImageConcatenateMulti_UTK**:智能拼接多张图像,支持2-4图自动布局 +  +-#### 图像变换与调整 +-- **ImageScaleByAspectRatio_UTK**:按指定宽高比缩放图像 +-- **ImageMaskScaleAs_UTK**:按参考图像尺寸缩放图像 ++- #### 图像变换与调整 ++- **ResizeImageVerKJ_UTK**:KJ v2 风格的高兼容缩放,支持 stretch/resize/pad/pad_edge/pad_edge_pixel/crop/pillarbox_blur/total_pixels 与 `crop_position` ++- **ImageScaleByAspectRatio_UTK**:按指定宽高比缩放图像(已支持与 KJ v2 一致的 fit 模式与 `crop_position`,背景色为预设清单) ++- **ImageMaskScaleAs_UTK**:按参考图像尺寸缩放图像(已支持与 KJ v2 一致的 fit 模式与 `crop_position`,pad_color 为预设清单) + - **ImageScaleRestore_UTK**:将图像恢复到原始尺寸 + - **ImageRemoveAlpha_UTK**:移除图像的Alpha通道 + - **ImageCombineAlpha_UTK**:合并Alpha通道到图像 +@@ -81,6 +82,7 @@ tqdm + - **MaskAnd_UTK**:掩码与运算 + - **MaskSub_UTK**:掩码减法运算 + - **MaskAdd_UTK**:掩码加法运算 ++- **BlockifyMask_UTK**:将掩码按 block_size 马赛克化(支持 cpu/cuda;可选二值化) +  + ### 🛠️ 工具节点 +  +@@ -91,6 +93,7 @@ tqdm +  + #### 数学与逻辑 + - **MathExpression_UTK**:数学表达式计算,支持复杂公式和函数 ++- **BestContextWindow_UTK**:最佳滑动窗口帧数计算(满足 4n+1,最小化补帧;输出 best_window/padding/padded_total/segments) +  + #### 系统工具 + - **PurgeVRAM_UTK**:显存清理,支持选择性清理缓存和模型 +@@ -176,7 +179,22 @@ AudioCropProcess_UTK +  + ## 📋 版本历史 +  +-### v1.3.2 (最新) ++### v1.4.7 (最新) ++- 修复 resize 与 pad 方法表现相同的问题: ++ - ResizeImageVerKJ (UTK):resize 模式只等比缩放不填充,pad 模式填充到目标尺寸 ++ - ImageMaskScaleAs (UTK):resize 返回实际缩放尺寸,pad 填充到目标尺寸并正确输出尺寸 ++ - ImageScaleByAspectRatio (UTK):resize 返回实际缩放尺寸,pad 填充到目标尺寸并正确输出尺寸 ++ - resize:等比缩放,输出尺寸 = 缩放后尺寸(可能小于目标尺寸) ++ - pad:等比缩放 + 背景填充,输出尺寸 = 目标尺寸(固定尺寸) ++ ++### v1.4.6 ++- 新增 `Resize Image ver KJ (UTK)`,完整对齐 KJ v2 调整模式,支持 `crop_position` 与 mask 同步缩放;pad_edge/pad_edge_pixel 行为与 KJ 对齐 ++- 升级 `Image Mask Scale As (UTK)` 与 `Image Scale By Aspect Ratio (UTK)`:支持同样的 fit 模式、`crop_position`,并将背景色改为预设清单 ++- 新增 `Blockify Mask (UTK)`:掩码块化,支持二值化 ++- 新增 `Best Context Window (UTK)`:计算满足 4n+1 的最佳窗口,最小化补帧 ++- 统一分类命名:`UniversalToolkit/Tools` ++ ++### v1.3.2 + - 新增电商应用类,重新组织预设分类结构 + - 创建专门的电商应用类,包含6个专业电商功能: + - Ecommerce-Professional Product Photography (专业产品图) +diff --git a/__init__.py b/__init__.py +index a72fc01..2b6dda9 100644 +--- a/__init__.py ++++ b/__init__.py +@@ -8,13 +8,43 @@ A comprehensive toolkit for ComfyUI that provides various utility nodes for imag + :license: MIT, see LICENSE for more details. + """ +  +-__version__ = "1.4.4" ++__version__ = "1.4.7" + __author__ = "CyberDickLang" + __email__ = "286878701@qq.com" + __url__ = "https://github.com/whmc76" +  + # 更新日志 + CHANGELOG = { ++ "1.4.7": [ ++ "修复 resize 与 pad 方法表现相同的问题:", ++ "- ResizeImageVerKJ (UTK):resize 模式只等比缩放不填充,pad 模式填充到目标尺寸", ++ "- ImageMaskScaleAs (UTK):resize 返回实际缩放尺寸,pad 填充到目标尺寸并正确输出尺寸", ++ "- ImageScaleByAspectRatio (UTK):resize 返回实际缩放尺寸,pad 填充到目标尺寸并正确输出尺寸", ++ "- resize:等比缩放,输出尺寸 = 缩放后尺寸(可能小于目标尺寸)", ++ "- pad:等比缩放 + 背景填充,输出尺寸 = 目标尺寸(固定尺寸)", ++ ], ++ "1.4.6": [ ++ "新增 Resize Image ver KJ (UTK):", ++ "- 复刻 KJ v2 的调整模式:stretch/resize/pad/pad_edge/pad_edge_pixel/crop/pillarbox_blur/total_pixels", ++ "- 支持 mask 同步缩放与对齐,pad_edge/pad_edge_pixel 行为与 KJ 对齐", ++ "升级 Image Mask Scale As (UTK):", ++ "- fit 与 KJ v2 对齐,新增 crop_position,支持预设 pad_color", ++ "升级 Image Scale By Aspect Ratio (UTK):", ++ "- fit 与 KJ v2 对齐,新增 crop_position,background_color 改为预设清单", ++ "修正 pad_edge 与 pad_edge_pixel 的边缘与角点处理逻辑,匹配 KJ 视觉表现", ++ ], ++ "1.4.5": [ ++ "新增Best Context Window (UTK)节点:", ++ "- 计算满足4n+1且位于[min,max]区间的最佳窗口,以最小化补帧", ++ "- 输出best_window、padding、padded_total、segments", ++ "- 分类:UniversalToolkit/Tools", ++ "新增Blockify Mask (UTK)节点:", ++ "- 将掩码按block_size块化,支持cpu/cuda", ++ "- 可选二值化binarize与threshold", ++ "- 分类:UniversalToolkit/Mask", ++ "统一分类命名:将UniversalToolkit/tools合并为UniversalToolkit/Tools", ++ "修复:Get Image or Mask Range From Batch (UTK) 分类名不一致问题", ++ ], + "1.4.4": [ + "新增Get Image or Mask Range From Batch (UTK)节点:", + "- 支持从图像批次或遮罩批次中提取指定范围的元素", +@@ -742,9 +772,27 @@ try: + NODE_CLASS_MAPPINGS as GET_IMAGE_RANGE_MAPPINGS + from .nodes.tools.get_image_range_from_batch import \ + NODE_DISPLAY_NAME_MAPPINGS as GET_IMAGE_RANGE_DISPLAY ++ from .nodes.tools.optimal_context_window_node import \ ++ NODE_CLASS_MAPPINGS as BEST_CONTEXT_WINDOW_MAPPINGS ++ from .nodes.tools.optimal_context_window_node import \ ++ NODE_DISPLAY_NAME_MAPPINGS as BEST_CONTEXT_WINDOW_DISPLAY ++ from .nodes.mask.blockify_mask import \ ++ NODE_CLASS_MAPPINGS as BLOCKIFY_MASK_MAPPINGS ++ from .nodes.mask.blockify_mask import \ ++ NODE_DISPLAY_NAME_MAPPINGS as BLOCKIFY_MASK_DISPLAY ++ from .nodes.image.resize_image_ver_kj import \ ++ NODE_CLASS_MAPPINGS as RESIZE_VER_KJ_MAPPINGS ++ from .nodes.image.resize_image_ver_kj import \ ++ NODE_DISPLAY_NAME_MAPPINGS as RESIZE_VER_KJ_DISPLAY + except ImportError: + GET_IMAGE_RANGE_MAPPINGS = {} + GET_IMAGE_RANGE_DISPLAY = {} ++ BEST_CONTEXT_WINDOW_MAPPINGS = {} ++ BEST_CONTEXT_WINDOW_DISPLAY = {} ++ BLOCKIFY_MASK_MAPPINGS = {} ++ BLOCKIFY_MASK_DISPLAY = {} ++ RESIZE_VER_KJ_MAPPINGS = {} ++ RESIZE_VER_KJ_DISPLAY = {} +  +  + # 合并所有节点映射 +@@ -786,6 +834,9 @@ NODE_CLASS_MAPPINGS.update(COLOR_TO_MASK_MAPPINGS) + NODE_CLASS_MAPPINGS.update(LAZY_SWITCH_MAPPINGS) + NODE_CLASS_MAPPINGS.update(TEXT_TRANSLATOR_API_MAPPINGS) + NODE_CLASS_MAPPINGS.update(GET_IMAGE_RANGE_MAPPINGS) ++NODE_CLASS_MAPPINGS.update(BEST_CONTEXT_WINDOW_MAPPINGS) ++NODE_CLASS_MAPPINGS.update(BLOCKIFY_MASK_MAPPINGS) ++NODE_CLASS_MAPPINGS.update(RESIZE_VER_KJ_MAPPINGS) +  + # 合并显示名称映射 + NODE_DISPLAY_NAME_MAPPINGS = {} +@@ -826,6 +877,9 @@ NODE_DISPLAY_NAME_MAPPINGS.update(COLOR_TO_MASK_DISPLAY) + NODE_DISPLAY_NAME_MAPPINGS.update(LAZY_SWITCH_DISPLAY) + NODE_DISPLAY_NAME_MAPPINGS.update(TEXT_TRANSLATOR_API_DISPLAY) + NODE_DISPLAY_NAME_MAPPINGS.update(GET_IMAGE_RANGE_DISPLAY) ++NODE_DISPLAY_NAME_MAPPINGS.update(BEST_CONTEXT_WINDOW_DISPLAY) ++NODE_DISPLAY_NAME_MAPPINGS.update(BLOCKIFY_MASK_DISPLAY) ++NODE_DISPLAY_NAME_MAPPINGS.update(RESIZE_VER_KJ_DISPLAY) +  + NODE_CATEGORIES = { + "UniversalToolkit": [ +@@ -867,6 +921,9 @@ NODE_CATEGORIES = { + "APIImageGenerator_UTK", + "TextTranslatorAPI_UTK", + "GetImageRangeFromBatch_UTK", ++ "BestContextWindow_UTK", ++ "BlockifyMask_UTK", ++ "ResizeImageVerKJ_UTK", + ] + } +  +diff --git a/nodes/image/image_mask_scale_as.py b/nodes/image/image_mask_scale_as.py +index 7debdaa..46af644 100644 +--- a/nodes/image/image_mask_scale_as.py ++++ b/nodes/image/image_mask_scale_as.py +@@ -9,7 +9,7 @@ Scales images and masks to match the dimensions of a reference image. + """ +  + import torch +-from PIL import Image ++from PIL import Image, ImageFilter +  + from ..image_utils import image2mask, pil2tensor, tensor2pil +  +@@ -32,10 +32,18 @@ def fit_resize_image( + target_height, + fit_mode, + resize_sampler, +- background_color="#000000", ++ background_color="black", ++ crop_position="center", + ): + """Resize image according to fit mode""" +- if fit_mode == "letterbox": ++ if fit_mode == "resize": ++ # resize: 只等比缩放,不填充,直接返回缩放后的图像 ++ scale = min(target_width / image.width, target_height / image.height) ++ new_width = int(image.width * scale) ++ new_height = int(image.height * scale) ++ return image.resize((new_width, new_height), resize_sampler) ++  ++ if fit_mode in ["letterbox", "pad", "pad_edge", "pad_edge_pixel", "pillarbox_blur"]: + # Calculate scaling factor to fit within target dimensions + scale = min(target_width / image.width, target_height / image.height) + new_width = int(image.width * scale) +@@ -44,13 +52,164 @@ def fit_resize_image( + # Resize image + resized = image.resize((new_width, new_height), resize_sampler) +  +- # Create new image with target dimensions and paste resized image +- if image.mode == "RGB": +- result = Image.new("RGB", (target_width, target_height), background_color) ++ # Create background ++ if fit_mode == "pillarbox_blur": ++ # create scaled background then blur and dim ++ scale_fill = max(target_width / max(1, image.width), target_height / max(1, image.height)) ++ bg_w = max(1, int(round(image.width * scale_fill))) ++ bg_h = max(1, int(round(image.height * scale_fill))) ++ bg = image.resize((bg_w, bg_h), Image.BILINEAR) ++ # center crop to canvas ++ x0 = max(0, (bg_w - target_width) // 2) ++ y0 = max(0, (bg_h - target_height) // 2) ++ bg = bg.crop((x0, y0, x0 + target_width, y0 + target_height)) ++ sigma = max(1.0, 0.006 * float(min(target_width, target_height))) ++ bg = bg.filter(ImageFilter.GaussianBlur(radius=sigma)) ++ # desaturate slightly if RGB ++ if bg.mode == "RGB": ++ r, g, b = bg.split() ++ # simple luminance ++ l = r.point(lambda v: int(0.2126 * v)) ++ l = Image.merge("RGB", (l, l, l)) ++ def mix(a, b, t=0.2): ++ return Image.blend(a, b, t) ++ bg = mix(bg, l) ++ # dim ++ bg = bg.point(lambda v: int(v * 0.35)) ++ result = bg ++ elif fit_mode in ["pad_edge", "pad_edge_pixel"]: ++ # start with empty canvas ++ result = Image.new("RGB" if image.mode == "RGB" else "L", (target_width, target_height)) + else: +- result = Image.new("L", (target_width, target_height), 0) +- paste_x = (target_width - new_width) // 2 +- paste_y = (target_height - new_height) // 2 ++ if image.mode == "RGB": ++ # preset color names ++ preset_colors = { ++ "black": "#000000", ++ "white": "#FFFFFF", ++ "gray": "#808080", ++ "red": "#FF0000", ++ "green": "#00FF00", ++ "blue": "#0000FF", ++ "yellow": "#FFFF00", ++ "cyan": "#00FFFF", ++ "magenta": "#FF00FF", ++ } ++ fill_color = preset_colors.get(str(background_color).lower(), background_color) ++ result = Image.new("RGB", (target_width, target_height), fill_color) ++ else: ++ result = Image.new("L", (target_width, target_height), 0) ++ ++ # paste location ++ if crop_position == "center": ++ paste_x = (target_width - new_width) // 2 ++ paste_y = (target_height - new_height) // 2 ++ elif crop_position == "top": ++ paste_x = (target_width - new_width) // 2 ++ paste_y = 0 ++ elif crop_position == "bottom": ++ paste_x = (target_width - new_width) // 2 ++ paste_y = target_height - new_height ++ elif crop_position == "left": ++ paste_x = 0 ++ paste_y = (target_height - new_height) // 2 ++ elif crop_position == "right": ++ paste_x = target_width - new_width ++ paste_y = (target_height - new_height) // 2 ++ else: ++ paste_x = (target_width - new_width) // 2 ++ paste_y = (target_height - new_height) // 2 ++ # specialized edge padding behaviors ++ if fit_mode == "pad_edge" or fit_mode == "pad_edge_pixel": ++ left_pad = paste_x ++ right_pad = target_width - (paste_x + new_width) ++ top_pad = paste_y ++ bottom_pad = target_height - (paste_y + new_height) ++ ++ # left/right stripes from image columns ++ if left_pad > 0: ++ col = resized.crop((0, 0, 1, new_height)) ++ if fit_mode == "pad_edge_pixel": ++ col = col.resize((left_pad, new_height), Image.NEAREST) ++ result.paste(col, (0, paste_y)) ++ else: ++ # mean color of left edge ++ if col.mode == "RGB": ++ pixels = list(col.getdata()) ++ r = sum(p[0] for p in pixels) // len(pixels) ++ g = sum(p[1] for p in pixels) // len(pixels) ++ b = sum(p[2] for p in pixels) // len(pixels) ++ fill = (r, g, b) ++ else: ++ v = sum(col.getdata()) // len(col.getdata()) ++ fill = v ++ Image.Image.paste(result, Image.new(result.mode, (left_pad, new_height), fill), (0, paste_y)) ++ ++ if right_pad > 0: ++ col = resized.crop((new_width - 1, 0, new_width, new_height)) ++ if fit_mode == "pad_edge_pixel": ++ col = col.resize((right_pad, new_height), Image.NEAREST) ++ result.paste(col, (paste_x + new_width, paste_y)) ++ else: ++ if col.mode == "RGB": ++ pixels = list(col.getdata()) ++ r = sum(p[0] for p in pixels) // len(pixels) ++ g = sum(p[1] for p in pixels) // len(pixels) ++ b = sum(p[2] for p in pixels) // len(pixels) ++ fill = (r, g, b) ++ else: ++ v = sum(col.getdata()) // len(col.getdata()) ++ fill = v ++ Image.Image.paste(result, Image.new(result.mode, (right_pad, new_height), fill), (paste_x + new_width, paste_y)) ++ ++ # top/bottom stripes from image rows ++ if top_pad > 0: ++ row = resized.crop((0, 0, new_width, 1)) ++ if fit_mode == "pad_edge_pixel": ++ row = row.resize((new_width, top_pad), Image.NEAREST) ++ result.paste(row, (paste_x, 0)) ++ # corners by corner pixels ++ if left_pad > 0: ++ c = resized.getpixel((0, 0)) ++ Image.Image.paste(result, Image.new(result.mode, (left_pad, top_pad), c), (0, 0)) ++ if right_pad > 0: ++ c = resized.getpixel((new_width - 1, 0 \ No newline at end of file diff --git a/nodes/image/image_mask_scale_as.py b/nodes/image/image_mask_scale_as.py index 7debdaa..46af644 100644 --- a/nodes/image/image_mask_scale_as.py +++ b/nodes/image/image_mask_scale_as.py @@ -9,7 +9,7 @@ Scales images and masks to match the dimensions of a reference image. """ import torch -from PIL import Image +from PIL import Image, ImageFilter from ..image_utils import image2mask, pil2tensor, tensor2pil @@ -32,10 +32,18 @@ def fit_resize_image( target_height, fit_mode, resize_sampler, - background_color="#000000", + background_color="black", + crop_position="center", ): """Resize image according to fit mode""" - if fit_mode == "letterbox": + if fit_mode == "resize": + # resize: 只等比缩放,不填充,直接返回缩放后的图像 + scale = min(target_width / image.width, target_height / image.height) + new_width = int(image.width * scale) + new_height = int(image.height * scale) + return image.resize((new_width, new_height), resize_sampler) + + if fit_mode in ["letterbox", "pad", "pad_edge", "pad_edge_pixel", "pillarbox_blur"]: # Calculate scaling factor to fit within target dimensions scale = min(target_width / image.width, target_height / image.height) new_width = int(image.width * scale) @@ -44,13 +52,164 @@ def fit_resize_image( # Resize image resized = image.resize((new_width, new_height), resize_sampler) - # Create new image with target dimensions and paste resized image - if image.mode == "RGB": - result = Image.new("RGB", (target_width, target_height), background_color) + # Create background + if fit_mode == "pillarbox_blur": + # create scaled background then blur and dim + scale_fill = max(target_width / max(1, image.width), target_height / max(1, image.height)) + bg_w = max(1, int(round(image.width * scale_fill))) + bg_h = max(1, int(round(image.height * scale_fill))) + bg = image.resize((bg_w, bg_h), Image.BILINEAR) + # center crop to canvas + x0 = max(0, (bg_w - target_width) // 2) + y0 = max(0, (bg_h - target_height) // 2) + bg = bg.crop((x0, y0, x0 + target_width, y0 + target_height)) + sigma = max(1.0, 0.006 * float(min(target_width, target_height))) + bg = bg.filter(ImageFilter.GaussianBlur(radius=sigma)) + # desaturate slightly if RGB + if bg.mode == "RGB": + r, g, b = bg.split() + # simple luminance + l = r.point(lambda v: int(0.2126 * v)) + l = Image.merge("RGB", (l, l, l)) + def mix(a, b, t=0.2): + return Image.blend(a, b, t) + bg = mix(bg, l) + # dim + bg = bg.point(lambda v: int(v * 0.35)) + result = bg + elif fit_mode in ["pad_edge", "pad_edge_pixel"]: + # start with empty canvas + result = Image.new("RGB" if image.mode == "RGB" else "L", (target_width, target_height)) else: - result = Image.new("L", (target_width, target_height), 0) - paste_x = (target_width - new_width) // 2 - paste_y = (target_height - new_height) // 2 + if image.mode == "RGB": + # preset color names + preset_colors = { + "black": "#000000", + "white": "#FFFFFF", + "gray": "#808080", + "red": "#FF0000", + "green": "#00FF00", + "blue": "#0000FF", + "yellow": "#FFFF00", + "cyan": "#00FFFF", + "magenta": "#FF00FF", + } + fill_color = preset_colors.get(str(background_color).lower(), background_color) + result = Image.new("RGB", (target_width, target_height), fill_color) + else: + result = Image.new("L", (target_width, target_height), 0) + + # paste location + if crop_position == "center": + paste_x = (target_width - new_width) // 2 + paste_y = (target_height - new_height) // 2 + elif crop_position == "top": + paste_x = (target_width - new_width) // 2 + paste_y = 0 + elif crop_position == "bottom": + paste_x = (target_width - new_width) // 2 + paste_y = target_height - new_height + elif crop_position == "left": + paste_x = 0 + paste_y = (target_height - new_height) // 2 + elif crop_position == "right": + paste_x = target_width - new_width + paste_y = (target_height - new_height) // 2 + else: + paste_x = (target_width - new_width) // 2 + paste_y = (target_height - new_height) // 2 + # specialized edge padding behaviors + if fit_mode == "pad_edge" or fit_mode == "pad_edge_pixel": + left_pad = paste_x + right_pad = target_width - (paste_x + new_width) + top_pad = paste_y + bottom_pad = target_height - (paste_y + new_height) + + # left/right stripes from image columns + if left_pad > 0: + col = resized.crop((0, 0, 1, new_height)) + if fit_mode == "pad_edge_pixel": + col = col.resize((left_pad, new_height), Image.NEAREST) + result.paste(col, (0, paste_y)) + else: + # mean color of left edge + if col.mode == "RGB": + pixels = list(col.getdata()) + r = sum(p[0] for p in pixels) // len(pixels) + g = sum(p[1] for p in pixels) // len(pixels) + b = sum(p[2] for p in pixels) // len(pixels) + fill = (r, g, b) + else: + v = sum(col.getdata()) // len(col.getdata()) + fill = v + Image.Image.paste(result, Image.new(result.mode, (left_pad, new_height), fill), (0, paste_y)) + + if right_pad > 0: + col = resized.crop((new_width - 1, 0, new_width, new_height)) + if fit_mode == "pad_edge_pixel": + col = col.resize((right_pad, new_height), Image.NEAREST) + result.paste(col, (paste_x + new_width, paste_y)) + else: + if col.mode == "RGB": + pixels = list(col.getdata()) + r = sum(p[0] for p in pixels) // len(pixels) + g = sum(p[1] for p in pixels) // len(pixels) + b = sum(p[2] for p in pixels) // len(pixels) + fill = (r, g, b) + else: + v = sum(col.getdata()) // len(col.getdata()) + fill = v + Image.Image.paste(result, Image.new(result.mode, (right_pad, new_height), fill), (paste_x + new_width, paste_y)) + + # top/bottom stripes from image rows + if top_pad > 0: + row = resized.crop((0, 0, new_width, 1)) + if fit_mode == "pad_edge_pixel": + row = row.resize((new_width, top_pad), Image.NEAREST) + result.paste(row, (paste_x, 0)) + # corners by corner pixels + if left_pad > 0: + c = resized.getpixel((0, 0)) + Image.Image.paste(result, Image.new(result.mode, (left_pad, top_pad), c), (0, 0)) + if right_pad > 0: + c = resized.getpixel((new_width - 1, 0)) + Image.Image.paste(result, Image.new(result.mode, (right_pad, top_pad), c), (paste_x + new_width, 0)) + else: + if row.mode == "RGB": + pixels = list(row.getdata()) + r = sum(p[0] for p in pixels) // len(pixels) + g = sum(p[1] for p in pixels) // len(pixels) + b = sum(p[2] for p in pixels) // len(pixels) + fill = (r, g, b) + else: + v = sum(row.getdata()) // len(row.getdata()) + fill = v + Image.Image.paste(result, Image.new(result.mode, (target_width, top_pad), fill), (0, 0)) + + if bottom_pad > 0: + row = resized.crop((0, new_height - 1, new_width, new_height)) + if fit_mode == "pad_edge_pixel": + row = row.resize((new_width, bottom_pad), Image.NEAREST) + result.paste(row, (paste_x, paste_y + new_height)) + if left_pad > 0: + c = resized.getpixel((0, new_height - 1)) + Image.Image.paste(result, Image.new(result.mode, (left_pad, bottom_pad), c), (0, paste_y + new_height)) + if right_pad > 0: + c = resized.getpixel((new_width - 1, new_height - 1)) + Image.Image.paste(result, Image.new(result.mode, (right_pad, bottom_pad), c), (paste_x + new_width, paste_y + new_height)) + else: + if row.mode == "RGB": + pixels = list(row.getdata()) + r = sum(p[0] for p in pixels) // len(pixels) + g = sum(p[1] for p in pixels) // len(pixels) + b = sum(p[2] for p in pixels) // len(pixels) + fill = (r, g, b) + else: + v = sum(row.getdata()) // len(row.getdata()) + fill = v + Image.Image.paste(result, Image.new(result.mode, (target_width, bottom_pad), fill), (0, paste_y + new_height)) + + # finally paste the resized content result.paste(resized, (paste_x, paste_y)) return result @@ -64,13 +223,29 @@ def fit_resize_image( resized = image.resize((new_width, new_height), resize_sampler) # Crop to target dimensions - crop_x = (new_width - target_width) // 2 - crop_y = (new_height - target_height) // 2 + if crop_position == "center": + crop_x = (new_width - target_width) // 2 + crop_y = (new_height - target_height) // 2 + elif crop_position == "top": + crop_x = (new_width - target_width) // 2 + crop_y = 0 + elif crop_position == "bottom": + crop_x = (new_width - target_width) // 2 + crop_y = new_height - target_height + elif crop_position == "left": + crop_x = 0 + crop_y = (new_height - target_height) // 2 + elif crop_position == "right": + crop_x = new_width - target_width + crop_y = (new_height - target_height) // 2 + else: + crop_x = (new_width - target_width) // 2 + crop_y = (new_height - target_height) // 2 return resized.crop( (crop_x, crop_y, crop_x + target_width, crop_y + target_height) ) - else: # fill + else: # stretch/fill # Simple resize to target dimensions return image.resize((target_width, target_height), resize_sampler) @@ -80,7 +255,15 @@ class ImageMaskScaleAs_UTK: @classmethod def INPUT_TYPES(cls): - fit_mode = ["letterbox", "crop", "fill"] + fit_mode = [ + "stretch", + "resize", + "pad", + "pad_edge", + "pad_edge_pixel", + "crop", + "pillarbox_blur", + ] method_mode = ["lanczos", "bicubic", "hamming", "bilinear", "box", "nearest"] return { @@ -92,6 +275,18 @@ class ImageMaskScaleAs_UTK: "optional": { "image": ("IMAGE",), # "mask": ("MASK",), # + "pad_color": ([ + "black", + "white", + "gray", + "red", + "green", + "blue", + "yellow", + "cyan", + "magenta", + ], {"default": "black"}), + "crop_position": (["center", "top", "bottom", "left", "right"], {"default": "center"}), }, } @@ -112,6 +307,8 @@ class ImageMaskScaleAs_UTK: method, image=None, mask=None, + pad_color="black", + crop_position="center", ): if scale_as.shape[0] > 0: _asimage = tensor2pil(scale_as[0]) @@ -137,14 +334,20 @@ class ImageMaskScaleAs_UTK: ret_images = [] ret_masks = [] + output_width = target_width + output_height = target_height + if image is not None: for i in image: i = torch.unsqueeze(i, 0) _image = tensor2pil(i).convert("RGB") orig_width, orig_height = _image.size _image = fit_resize_image( - _image, target_width, target_height, fit, resize_sampler + _image, target_width, target_height, fit, resize_sampler, pad_color, crop_position ) + # For resize mode, use actual image size instead of target size + if fit == "resize": + output_width, output_height = _image.size ret_images.append(pil2tensor(_image)) if mask is not None: if mask.dim() == 2: @@ -153,9 +356,13 @@ class ImageMaskScaleAs_UTK: m = torch.unsqueeze(m, 0) _mask = tensor2pil(m).convert("L") orig_width, orig_height = _mask.size + # Mask padding背景始终为黑 _mask = fit_resize_image( - _mask, target_width, target_height, fit, resize_sampler + _mask, target_width, target_height, fit, resize_sampler, "#000000", crop_position ).convert("L") + # For resize mode, use actual mask size instead of target size + if fit == "resize": + output_width, output_height = _mask.size ret_masks.append(image2mask(_mask)) if len(ret_images) > 0 and len(ret_masks) > 0: log( @@ -166,8 +373,8 @@ class ImageMaskScaleAs_UTK: torch.cat(ret_images, dim=0), torch.cat(ret_masks, dim=0), [orig_width, orig_height], - target_width, - target_height, + output_width, + output_height, ) elif len(ret_images) > 0 and len(ret_masks) == 0: log( @@ -178,8 +385,8 @@ class ImageMaskScaleAs_UTK: torch.cat(ret_images, dim=0), None, [orig_width, orig_height], - target_width, - target_height, + output_width, + output_height, ) elif len(ret_images) == 0 and len(ret_masks) > 0: log( @@ -190,8 +397,8 @@ class ImageMaskScaleAs_UTK: None, torch.cat(ret_masks, dim=0), [orig_width, orig_height], - target_width, - target_height, + output_width, + output_height, ) else: log( diff --git a/nodes/image/image_scale_by_aspect_ratio.py b/nodes/image/image_scale_by_aspect_ratio.py index fce8622..1357e6d 100644 --- a/nodes/image/image_scale_by_aspect_ratio.py +++ b/nodes/image/image_scale_by_aspect_ratio.py @@ -11,7 +11,7 @@ Scales images to specific aspect ratios with various fitting modes. import math import torch -from PIL import Image +from PIL import Image, ImageFilter from ..image_utils import (fit_resize_image, image2mask, is_valid_mask, log, num_round_up_to_multiple, pil2tensor, tensor2pil) @@ -35,10 +35,23 @@ def num_round_up_to_multiple(num, multiple): def fit_resize_image( - image, target_width, target_height, fit_mode, resize_sampler, background_color + image, + target_width, + target_height, + fit_mode, + resize_sampler, + background_color, + crop_position="center", ): """Resize image according to fit mode""" - if fit_mode == "letterbox": + if fit_mode == "resize": + # resize: 只等比缩放,不填充,直接返回缩放后的图像 + scale = min(target_width / image.width, target_height / image.height) + new_width = int(image.width * scale) + new_height = int(image.height * scale) + return image.resize((new_width, new_height), resize_sampler) + + if fit_mode in ["letterbox", "pad", "pad_edge", "pad_edge_pixel", "pillarbox_blur"]: # Calculate scaling factor to fit within target dimensions scale = min(target_width / image.width, target_height / image.height) new_width = int(image.width * scale) @@ -47,10 +60,127 @@ def fit_resize_image( # Resize image resized = image.resize((new_width, new_height), resize_sampler) - # Create new image with target dimensions and paste resized image - result = Image.new(image.mode, (target_width, target_height), background_color) - paste_x = (target_width - new_width) // 2 - paste_y = (target_height - new_height) // 2 + # Create background + if fit_mode == "pillarbox_blur": + scale_fill = max(target_width / max(1, image.width), target_height / max(1, image.height)) + bg_w = max(1, int(round(image.width * scale_fill))) + bg_h = max(1, int(round(image.height * scale_fill))) + bg = image.resize((bg_w, bg_h), Image.BILINEAR) + x0 = max(0, (bg_w - target_width) // 2) + y0 = max(0, (bg_h - target_height) // 2) + bg = bg.crop((x0, y0, x0 + target_width, y0 + target_height)) + sigma = max(1.0, 0.006 * float(min(target_width, target_height))) + bg = bg.filter(ImageFilter.GaussianBlur(radius=sigma)) + if bg.mode == "RGB": + r, g, b = bg.split() + l = r.point(lambda v: int(0.2126 * v)) + l = Image.merge("RGB", (l, l, l)) + bg = Image.blend(bg, l, 0.2) + bg = bg.point(lambda v: int(v * 0.35)) + result = bg + else: + result = Image.new(image.mode, (target_width, target_height), background_color) + + # paste position + if crop_position == "center": + paste_x = (target_width - new_width) // 2 + paste_y = (target_height - new_height) // 2 + elif crop_position == "top": + paste_x = (target_width - new_width) // 2 + paste_y = 0 + elif crop_position == "bottom": + paste_x = (target_width - new_width) // 2 + paste_y = target_height - new_height + elif crop_position == "left": + paste_x = 0 + paste_y = (target_height - new_height) // 2 + elif crop_position == "right": + paste_x = target_width - new_width + paste_y = (target_height - new_height) // 2 + else: + paste_x = (target_width - new_width) // 2 + paste_y = (target_height - new_height) // 2 + + # Apply pad_edge / pad_edge_pixel stripes + if fit_mode in ["pad_edge", "pad_edge_pixel"]: + left_pad = paste_x + right_pad = target_width - (paste_x + new_width) + top_pad = paste_y + bottom_pad = target_height - (paste_y + new_height) + + if left_pad > 0: + col = resized.crop((0, 0, 1, new_height)) + if fit_mode == "pad_edge_pixel": + result.paste(col.resize((left_pad, new_height), Image.NEAREST), (0, paste_y)) + else: + if col.mode == "RGB": + pixels = list(col.getdata()) + r = sum(p[0] for p in pixels) // len(pixels) + g = sum(p[1] for p in pixels) // len(pixels) + b = sum(p[2] for p in pixels) // len(pixels) + fill = (r, g, b) + else: + v = sum(col.getdata()) // len(col.getdata()) + fill = v + Image.Image.paste(result, Image.new(result.mode, (left_pad, new_height), fill), (0, paste_y)) + if right_pad > 0: + col = resized.crop((new_width - 1, 0, new_width, new_height)) + if fit_mode == "pad_edge_pixel": + result.paste(col.resize((right_pad, new_height), Image.NEAREST), (paste_x + new_width, paste_y)) + else: + if col.mode == "RGB": + pixels = list(col.getdata()) + r = sum(p[0] for p in pixels) // len(pixels) + g = sum(p[1] for p in pixels) // len(pixels) + b = sum(p[2] for p in pixels) // len(pixels) + fill = (r, g, b) + else: + v = sum(col.getdata()) // len(col.getdata()) + fill = v + Image.Image.paste(result, Image.new(result.mode, (right_pad, new_height), fill), (paste_x + new_width, paste_y)) + if top_pad > 0: + row = resized.crop((0, 0, new_width, 1)) + if fit_mode == "pad_edge_pixel": + result.paste(row.resize((new_width, top_pad), Image.NEAREST), (paste_x, 0)) + if left_pad > 0: + c = resized.getpixel((0, 0)) + Image.Image.paste(result, Image.new(result.mode, (left_pad, top_pad), c), (0, 0)) + if right_pad > 0: + c = resized.getpixel((new_width - 1, 0)) + Image.Image.paste(result, Image.new(result.mode, (right_pad, top_pad), c), (paste_x + new_width, 0)) + else: + if row.mode == "RGB": + pixels = list(row.getdata()) + r = sum(p[0] for p in pixels) // len(pixels) + g = sum(p[1] for p in pixels) // len(pixels) + b = sum(p[2] for p in pixels) // len(pixels) + fill = (r, g, b) + else: + v = sum(row.getdata()) // len(row.getdata()) + fill = v + Image.Image.paste(result, Image.new(result.mode, (target_width, top_pad), fill), (0, 0)) + if bottom_pad > 0: + row = resized.crop((0, new_height - 1, new_width, new_height)) + if fit_mode == "pad_edge_pixel": + result.paste(row.resize((new_width, bottom_pad), Image.NEAREST), (paste_x, paste_y + new_height)) + if left_pad > 0: + c = resized.getpixel((0, new_height - 1)) + Image.Image.paste(result, Image.new(result.mode, (left_pad, bottom_pad), c), (0, paste_y + new_height)) + if right_pad > 0: + c = resized.getpixel((new_width - 1, new_height - 1)) + Image.Image.paste(result, Image.new(result.mode, (right_pad, bottom_pad), c), (paste_x + new_width, paste_y + new_height)) + else: + if row.mode == "RGB": + pixels = list(row.getdata()) + r = sum(p[0] for p in pixels) // len(pixels) + g = sum(p[1] for p in pixels) // len(pixels) + b = sum(p[2] for p in pixels) // len(pixels) + fill = (r, g, b) + else: + v = sum(row.getdata()) // len(row.getdata()) + fill = v + Image.Image.paste(result, Image.new(result.mode, (target_width, bottom_pad), fill), (0, paste_y + new_height)) + result.paste(resized, (paste_x, paste_y)) return result @@ -64,8 +194,24 @@ def fit_resize_image( resized = image.resize((new_width, new_height), resize_sampler) # Crop to target dimensions - crop_x = (new_width - target_width) // 2 - crop_y = (new_height - target_height) // 2 + if crop_position == "center": + crop_x = (new_width - target_width) // 2 + crop_y = (new_height - target_height) // 2 + elif crop_position == "top": + crop_x = (new_width - target_width) // 2 + crop_y = 0 + elif crop_position == "bottom": + crop_x = (new_width - target_width) // 2 + crop_y = new_height - target_height + elif crop_position == "left": + crop_x = 0 + crop_y = (new_height - target_height) // 2 + elif crop_position == "right": + crop_x = new_width - target_width + crop_y = (new_height - target_height) // 2 + else: + crop_x = (new_width - target_width) // 2 + crop_y = (new_height - target_height) // 2 return resized.crop( (crop_x, crop_y, crop_x + target_width, crop_y + target_height) ) @@ -91,7 +237,15 @@ class ImageScaleByAspectRatio_UTK: "3:4", "9:16", ] - fit_mode = ["letterbox", "crop", "fill"] + fit_mode = [ + "stretch", + "resize", + "pad", + "pad_edge", + "pad_edge_pixel", + "crop", + "pillarbox_blur", + ] method_mode = ["lanczos", "bicubic", "hamming", "bilinear", "box", "nearest"] multiple_list = ["8", "16", "32", "64", "128", "256", "512", "None"] scale_to_list = [ @@ -121,7 +275,18 @@ class ImageScaleByAspectRatio_UTK: "INT", {"default": 1024, "min": 4, "max": 1e8, "step": 1}, ), - "background_color": ("STRING", {"default": "#000000"}), + "background_color": ([ + "black", + "white", + "gray", + "red", + "green", + "blue", + "yellow", + "cyan", + "magenta", + ], {"default": "black"}), + "crop_position": (["center", "top", "bottom", "left", "right"], {"default": "center"}), }, "optional": { "image": ("IMAGE",), @@ -158,6 +323,7 @@ class ImageScaleByAspectRatio_UTK: scale_to_side, scale_to_length, background_color, + crop_position, image=None, mask=None, ): @@ -294,6 +460,9 @@ class ImageScaleByAspectRatio_UTK: elif method == "nearest": resize_sampler = Image.NEAREST + output_width = target_width + output_height = target_height + if len(orig_images) > 0: for i in orig_images: _image = tensor2pil(i).convert("RGB") @@ -304,7 +473,11 @@ class ImageScaleByAspectRatio_UTK: fit, resize_sampler, background_color, + crop_position, ) + # For resize mode, use actual image size instead of target size + if fit == "resize": + output_width, output_height = _image.size ret_images.append(pil2tensor(_image)) if len(orig_masks) > 0: for m in orig_masks: @@ -315,8 +488,12 @@ class ImageScaleByAspectRatio_UTK: target_height, fit, resize_sampler, - background_color, + "black", + crop_position, ).convert("L") + # For resize mode, use actual mask size instead of target size + if fit == "resize": + output_width, output_height = _mask.size ret_masks.append(image2mask(_mask)) if len(ret_images) > 0 and len(ret_masks) > 0: log( @@ -327,8 +504,8 @@ class ImageScaleByAspectRatio_UTK: torch.cat(ret_images, dim=0), torch.cat(ret_masks, dim=0), [orig_width, orig_height], - target_width, - target_height, + output_width, + output_height, len(ret_images), ) elif len(ret_images) > 0 and len(ret_masks) == 0: @@ -340,8 +517,8 @@ class ImageScaleByAspectRatio_UTK: torch.cat(ret_images, dim=0), None, [orig_width, orig_height], - target_width, - target_height, + output_width, + output_height, len(ret_images), ) elif len(ret_images) == 0 and len(ret_masks) > 0: @@ -353,8 +530,8 @@ class ImageScaleByAspectRatio_UTK: None, torch.cat(ret_masks, dim=0), [orig_width, orig_height], - target_width, - target_height, + output_width, + output_height, len(ret_masks), ) else: diff --git a/nodes/image/resize_image_ver_kj.py b/nodes/image/resize_image_ver_kj.py new file mode 100644 index 0000000..8832633 --- /dev/null +++ b/nodes/image/resize_image_ver_kj.py @@ -0,0 +1,264 @@ +import math +import torch +import torch.nn.functional as F + +from comfy import model_management +from comfy.utils import common_upscale + + +class ResizeImageVerKJ_UTK: + upscale_methods = ["nearest-exact", "bilinear", "area", "bicubic", "lanczos"] + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "image": ("IMAGE",), + "width": ("INT", {"default": 512, "min": 0, "max": 16384, "step": 1}), + "height": ("INT", {"default": 512, "min": 0, "max": 16384, "step": 1}), + "upscale_method": (cls.upscale_methods,), + "keep_proportion": ( + [ + "stretch", + "resize", + "pad", + "pad_edge", + "pad_edge_pixel", + "crop", + "pillarbox_blur", + "total_pixels", + ], + {"default": "resize"}, + ), + "pad_color": ("STRING", {"default": "0, 0, 0"}), + "crop_position": ("STRING", {"default": "center"}), + "divisible_by": ("INT", {"default": 2, "min": 0, "max": 512, "step": 1}), + }, + "optional": { + "mask": ("MASK",), + "device": (["cpu", "gpu"], {"default": "cpu"}), + }, + } + + RETURN_TYPES = ("IMAGE", "INT", "INT", "MASK") + RETURN_NAMES = ("IMAGE", "width", "height", "mask") + FUNCTION = "resize" + CATEGORY = "UniversalToolkit/Image" + + def _parse_color(self, s: str, dtype, device): + try: + vals = [int(x.strip()) for x in s.split(",")] + except Exception: + vals = [0, 0, 0] + if len(vals) == 1: + vals = vals * 3 + vals = [max(0, min(255, v)) / 255.0 for v in vals[:3]] + return torch.tensor(vals, dtype=dtype, device=device) + + def _compute_crop_rect(self, old_w, old_h, target_w, target_h, position: str): + old_aspect = old_w / old_h + new_aspect = target_w / target_h + if old_aspect > new_aspect: + crop_w = round(old_h * new_aspect) + crop_h = old_h + else: + crop_w = old_w + crop_h = round(old_w / new_aspect) + if position == "center": + x = (old_w - crop_w) // 2 + y = (old_h - crop_h) // 2 + elif position == "top": + x, y = (old_w - crop_w) // 2, 0 + elif position == "bottom": + x, y = (old_w - crop_w) // 2, old_h - crop_h + elif position == "left": + x, y = 0, (old_h - crop_h) // 2 + elif position == "right": + x, y = old_w - crop_w, (old_h - crop_h) // 2 + else: + x, y = (old_w - crop_w) // 2, (old_h - crop_h) // 2 + return x, y, crop_w, crop_h + + def resize(self, image: torch.Tensor, width: int, height: int, keep_proportion: str, upscale_method: str, + divisible_by: int, pad_color: str, crop_position: str, device: str = "cpu", mask: torch.Tensor = None): + B, H, W, C = image.shape + + if device == "gpu": + if upscale_method == "lanczos": + raise Exception("Lanczos is not supported on the GPU") + torch_device = model_management.get_torch_device() + else: + torch_device = torch.device("cpu") + + pillarbox_blur = keep_proportion == "pillarbox_blur" + + pad_left = pad_right = pad_top = pad_bottom = 0 + + if keep_proportion in ["resize", "total_pixels", "pad", "pad_edge", "pad_edge_pixel", "pillarbox_blur"]: + if keep_proportion == "total_pixels": + total_pixels = max(1, width * height) + aspect_ratio = W / H if H != 0 else 1.0 + new_h = int(math.sqrt(total_pixels / aspect_ratio)) + new_w = int(math.sqrt(total_pixels * aspect_ratio)) + elif width == 0 and height == 0: + new_w, new_h = W, H + elif width == 0 and height != 0: + ratio = height / H if H != 0 else 1.0 + new_w, new_h = round(W * ratio), height + elif height == 0 and width != 0: + ratio = width / W if W != 0 else 1.0 + new_w, new_h = width, round(H * ratio) + else: + ratio = min(width / W if W != 0 else 1.0, height / H if H != 0 else 1.0) + new_w, new_h = max(1, round(W * ratio)), max(1, round(H * ratio)) + + if keep_proportion in ["pad", "pad_edge", "pad_edge_pixel", "pillarbox_blur"]: + if crop_position == "center": + pad_left = (width - new_w) // 2 + pad_right = width - new_w - pad_left + pad_top = (height - new_h) // 2 + pad_bottom = height - new_h - pad_top + elif crop_position == "top": + pad_left = (width - new_w) // 2 + pad_right = width - new_w - pad_left + pad_top = 0 + pad_bottom = height - new_h + elif crop_position == "bottom": + pad_left = (width - new_w) // 2 + pad_right = width - new_w - pad_left + pad_top = height - new_h + pad_bottom = 0 + elif crop_position == "left": + pad_left = 0 + pad_right = width - new_w + pad_top = (height - new_h) // 2 + pad_bottom = height - new_h - pad_top + elif crop_position == "right": + pad_left = width - new_w + pad_right = 0 + pad_top = (height - new_h) // 2 + pad_bottom = height - new_h - pad_top + + width, height = new_w, new_h + else: + # stretch or crop path keeps requested width/height directly + if width == 0: + width = W + if height == 0: + height = H + + if divisible_by > 1: + width = width - (width % divisible_by) + height = height - (height % divisible_by) + + # Crop prior to resizing + x_in = image if image.device == torch_device else image.to(torch_device) + m_in = None if mask is None else (mask if mask.device == torch_device else mask.to(torch_device)) + + if keep_proportion == "crop": + x, y, cw, ch = self._compute_crop_rect(W, H, width, height, crop_position) + x_in = x_in.narrow(-2, x, cw).narrow(-3, y, ch) + if m_in is not None: + m_in = m_in.narrow(-1, x, cw).narrow(-2, y, ch) + + # Resize image and optional mask + out_img = common_upscale(x_in.movedim(-1, 1), width, height, upscale_method, crop="disabled").movedim(1, -1) + out_m = None + if m_in is not None: + if upscale_method == "lanczos": + out_m = common_upscale(m_in.unsqueeze(1).repeat(1, 3, 1, 1), width, height, upscale_method, crop="disabled").movedim(1, -1)[:, :, :, 0] + else: + out_m = common_upscale(m_in.unsqueeze(1), width, height, upscale_method, crop="disabled").squeeze(1) + + # resize mode: just return the resized image, no padding + if keep_proportion == "resize": + return (out_img.cpu(), out_img.shape[2], out_img.shape[1], out_m.cpu() if out_m is not None else torch.zeros(64, 64)) + + # Padding if requested + if (keep_proportion in ["pad", "pad_edge", "pad_edge_pixel", "pillarbox_blur"]) and (pad_left > 0 or pad_right > 0 or pad_top > 0 or pad_bottom > 0): + padded_w = width + pad_left + pad_right + padded_h = height + pad_top + pad_bottom + if divisible_by > 1: + w_rem = padded_w % divisible_by + h_rem = padded_h % divisible_by + if w_rem > 0: + pad_right += divisible_by - w_rem + if h_rem > 0: + pad_bottom += divisible_by - h_rem + padded_w = width + pad_left + pad_right + padded_h = height + pad_top + pad_bottom + + if keep_proportion == "pad_edge" or keep_proportion == "pad_edge_pixel": + # Build canvas and apply edge/edge_pixel logic similar to KJ implementation + canvas = torch.zeros((B, padded_h, padded_w, C), dtype=out_img.dtype, device=out_img.device) + for b in range(B): + # content + canvas[b, pad_top:pad_top+height, pad_left:pad_left+width, :] = out_img[b] + if keep_proportion == "pad_edge": + # mean color along edges + top_edge = out_img[b, 0, :, :] + bottom_edge = out_img[b, height-1, :, :] + left_edge = out_img[b, :, 0, :] + right_edge = out_img[b, :, width-1, :] + if pad_top > 0: + canvas[b, :pad_top, :, :] = top_edge.mean(dim=0) + if pad_bottom > 0: + canvas[b, pad_top+height:, :, :] = bottom_edge.mean(dim=0) + if pad_left > 0: + canvas[b, :, :pad_left, :] = left_edge.mean(dim=0) + if pad_right > 0: + canvas[b, :, pad_left+width:, :] = right_edge.mean(dim=0) + else: + # edge_pixel: extend exact edge rows/columns + if pad_top > 0: + row = out_img[b, 0:1, :, :].expand(pad_top, width, C) + canvas[b, :pad_top, pad_left:pad_left+width, :] = row + # corners + if pad_left > 0: + tl = out_img[b, 0, 0, :] + canvas[b, :pad_top, :pad_left, :] = tl + if pad_right > 0: + tr = out_img[b, 0, width-1, :] + canvas[b, :pad_top, pad_left+width:, :] = tr + if pad_bottom > 0: + row = out_img[b, height-1:height, :, :].expand(pad_bottom, width, C) + canvas[b, pad_top+height:, pad_left:pad_left+width, :] = row + if pad_left > 0: + bl = out_img[b, height-1, 0, :] + canvas[b, pad_top+height:, :pad_left, :] = bl + if pad_right > 0: + br = out_img[b, height-1, width-1, :] + canvas[b, pad_top+height:, pad_left+width:, :] = br + if pad_left > 0: + col = out_img[b, :, 0:1, :].expand(height, pad_left, C) + canvas[b, pad_top:pad_top+height, :pad_left, :] = col + if pad_right > 0: + col = out_img[b, :, width-1:width, :].expand(height, pad_right, C) + canvas[b, pad_top:pad_top+height, pad_left+width:, :] = col + out_img = canvas + if out_m is not None: + # replicate for mask to keep crisp edges + out_m = F.pad(out_m.unsqueeze(1), (pad_left, pad_right, pad_top, pad_bottom), mode="replicate").squeeze(1) + else: + # color padding + bg = self._parse_color(pad_color, out_img.dtype, out_img.device) + canvas = torch.zeros((B, padded_h, padded_w, C), dtype=out_img.dtype, device=out_img.device) + canvas[:, :, :, 0] = bg[0] + if C > 1: + canvas[:, :, :, 1] = bg[1] + if C > 2: + canvas[:, :, :, 2] = bg[2] + canvas[:, pad_top:pad_top+height, pad_left:pad_left+width, :] = out_img + out_img = canvas + if out_m is not None: + mcanvas = torch.zeros((B, padded_h, padded_w), dtype=out_img.dtype, device=out_img.device) + mcanvas[:, pad_top:pad_top+height, pad_left:pad_left+width] = out_m + out_m = mcanvas + + return (out_img.cpu(), out_img.shape[2], out_img.shape[1], out_m.cpu() if out_m is not None else torch.zeros(64, 64)) + + +NODE_CLASS_MAPPINGS = {"ResizeImageVerKJ_UTK": ResizeImageVerKJ_UTK} +NODE_DISPLAY_NAME_MAPPINGS = {"ResizeImageVerKJ_UTK": "Resize Image ver KJ (UTK)"} + + diff --git a/nodes/mask/blockify_mask.py b/nodes/mask/blockify_mask.py new file mode 100644 index 0000000..c78aebc --- /dev/null +++ b/nodes/mask/blockify_mask.py @@ -0,0 +1,60 @@ +import torch +import torch.nn.functional as F + + +class BlockifyMask_UTK: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "masks": ("MASK",), + "block_size": ("INT", {"default": 16, "min": 1, "max": 4096, "step": 1}), + "device": (["cpu", "cuda"], {"default": "cpu"}), + }, + "optional": { + # 可选二值化 + "binarize": ("BOOLEAN", {"default": True}), + "threshold": ("FLOAT", {"default": 0.5, "min": 0.0, "max": 1.0, "step": 0.01}), + } + } + + RETURN_TYPES = ("MASK",) + RETURN_NAMES = ("mask",) + FUNCTION = "blockify" + CATEGORY = "UniversalToolkit/Mask" + DESCRIPTION = "将连续掩码按 block_size 进行像素块化(马赛克化),可选二值化。" + + def blockify(self, masks: torch.Tensor, block_size: int, device: str, binarize: bool = True, threshold: float = 0.5): + mask_tensor = masks + if block_size <= 1: + out = torch.clamp(mask_tensor, 0.0, 1.0) + return (out,) + + # 选择设备(多数情况下 CPU 足够;如选 cuda 则尝试放到 GPU) + use_cuda = device == "cuda" and torch.cuda.is_available() + x_in = mask_tensor + if use_cuda: + x_in = x_in.to("cuda") + + # BxHxW -> Bx1xHxW for pooling + x = x_in.unsqueeze(1).contiguous() + + # 平均池化到较小网格;ceil 对齐,边缘使用对称填充避免尺寸不整除 + pooled = F.avg_pool2d(x, kernel_size=block_size, stride=block_size, ceil_mode=True) + + # 还原到原尺寸,使用最近邻形成块状 + out = F.interpolate(pooled, size=(mask_tensor.shape[1], mask_tensor.shape[2]), mode="nearest").squeeze(1) + + if binarize: + out = (out >= threshold).float() + + out = torch.clamp(out, 0.0, 1.0) + if use_cuda: + out = out.to("cpu") + return (out,) + + +NODE_CLASS_MAPPINGS = {"BlockifyMask_UTK": BlockifyMask_UTK} +NODE_DISPLAY_NAME_MAPPINGS = {"BlockifyMask_UTK": "Blockify Mask (UTK)"} + + diff --git a/nodes/tools/get_image_range_from_batch.py b/nodes/tools/get_image_range_from_batch.py index 8642f5c..1b7c372 100644 --- a/nodes/tools/get_image_range_from_batch.py +++ b/nodes/tools/get_image_range_from_batch.py @@ -11,7 +11,7 @@ class GetImageRangeFromBatch_UTK: RETURN_TYPES = ("IMAGE", "MASK") RETURN_NAMES = ("image", "mask") FUNCTION = "get_range_from_batch" - CATEGORY = "UniversalToolkit/tools" + CATEGORY = "UniversalToolkit/Tools" @classmethod def INPUT_TYPES(cls): diff --git a/nodes/tools/optimal_context_window_node.py b/nodes/tools/optimal_context_window_node.py new file mode 100644 index 0000000..2f5e472 --- /dev/null +++ b/nodes/tools/optimal_context_window_node.py @@ -0,0 +1,84 @@ +class BestContextWindow_UTK: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "total_frames": ("INT", {"default": 1, "min": 0}), + "min_window_frames": ("INT", {"default": 61, "min": 1}), + "max_window_frames": ("INT", {"default": 81, "min": 1}), + } + } + + RETURN_TYPES = ( + "INT", # best_window + "INT", # padding (冗余帧数) + "INT", # padded_total 实际处理帧数 + "INT", # segments 段数k + ) + RETURN_NAMES = ( + "best_window", + "padding", + "padded_total", + "segments", + ) + FUNCTION = "compute" + CATEGORY = "UniversalToolkit/Tools" + + @staticmethod + def _best_window(total_frames: int, min_window: int, max_window: int) -> tuple[int, int, int, int]: + # sanitize inputs + if max_window is None or max_window < 1: + max_window = 1 + if total_frames is None or total_frames < 0: + total_frames = 0 + if min_window is None or min_window < 1: + min_window = 1 + if max_window < min_window: + max_window = min_window + + # generate candidates that satisfy 4n+1 and within [min_window, max_window] + candidates = [] + # align start to first (4n+1) >= min_window + start = min_window if min_window % 4 == 1 else (min_window + (4 - ((min_window - 1) % 4) - 1)) + for w in range(start, max_window + 1): + if w % 4 == 1: + candidates.append(w) + + if not candidates: + # Fallback: choose closest valid 4n+1 not exceeding max_window + # compute nearest below max_window + w = max_window - ((max_window - 1) % 4) + if w < 1: + w = 1 + candidates = [w] + + # choose window minimizing padding to next multiple; tie -> larger window + def metrics(w: int): + k = (total_frames + w - 1) // w # ceil(total_frames / w) + padding = k * w - total_frames + padded_total = k * w + return padding, padded_total, k + + best_w = candidates[0] + best_pad, best_padded_total, best_k = metrics(best_w) + for w in candidates[1:]: + pad, padded_total, k = metrics(w) + if pad < best_pad or (pad == best_pad and w > best_w): + best_w, best_pad, best_padded_total, best_k = w, pad, padded_total, k + + return best_w, best_pad, best_padded_total, best_k + + def compute(self, total_frames: int, min_window_frames: int, max_window_frames: int): + best_w, padding, padded_total, k = self._best_window(int(total_frames), int(min_window_frames), int(max_window_frames)) + return (best_w, padding, padded_total, k) + + +NODE_CLASS_MAPPINGS = { + "BestContextWindow_UTK": BestContextWindow_UTK, +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "BestContextWindow_UTK": "Best Context Window (UTK)", +} + + diff --git a/pyproject.toml b/pyproject.toml index b77b0b5..d43e59d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "universaltoolkit" description = "A comprehensive toolkit based on ComfyUI, providing image, mask, audio, and tools nodes, fully modular and v3 compatible." -version = "1.4.4" +version = "1.4.8" license = {file = "LICENSE"} dependencies = [ "torch>=1.9.0",