diff --git a/README.md b/README.md index 9810393..5505081 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,14 @@ https://github.com/deepghs/imgutils 模型文件全部由该库提供和管理下载,会自动下载到 `HF_HOME` 环境变量指定的目录下 -因此需要在 ComfyUI 的启动项中设置该环境变量 + +**因此需要在 ComfyUI 的启动项中设置该环境变量** +如果你是官方包,可以在 .\run_nvidia_gpu.bat 的开头中添加: +```bat +set "CURDIR=%cd%" +set HF_HOME="%CURDIR%\(你想要的字目录)" +``` +秋叶包可以不用设,默认是下载到 `.cache\huggingface\hub\` 下面 Segment-Anything 模型需要手动下载到 @@ -20,7 +27,8 @@ Segment-Anything 模型需要手动下载到 节点介绍: -#### 检测节点 +#### BBox节点 +![alt text](md_img/bbox_nodes.png) - Imgutils Generic Detector - 支持多种 `imgutils` 提供的多种基于anime的检测模型 - `detection_type`:检测类型 @@ -32,14 +40,23 @@ Segment-Anything 模型需要手动下载到 - Mask to BBox 、 BBox to Mask - 用于 `Mask` 和 `BBox` 之间的转换 + - BBoxFilter - - 用于过滤 `BBox`,可以根据置信度、面积和 标签进行过滤 - - `labels`的值依照的是 `Imgutils Generic Detector` 输出的图片中bbox上标注的标签,可以输入多个,用逗号分隔 + - 用于过滤 `BBox`,可以根据置信度、面积和标签进行过滤 + - `labels`的值依照的是 `Imgutils Generic Detector` 的输出 `image with boxes` 中bbox上标注的标签,可以输入多个,用逗号分隔 + + #### segment-anything节点 -基本上抄 Impact-Pack 的 +基本上抄的 Impact-Pack + +![alt text](md_img/detailer_example.png) +- SAMPredictorNode +- SAMLoader for SAMPredictorNode + #### segment 节点 +![alt text](md_img/segment_nodes.png) - Imgutils Auto Segmenter - 仅能对图片进行 **前景和背景** 的分割 @@ -56,6 +73,8 @@ Segment-Anything 模型需要手动下载到 #### Mask处理节点 收录一些常用的 Mask 处理节点 +![alt text](md_img/mask_nodes.png) + - Mask Morphology: - 提供了常用的形态学操作:膨胀、腐蚀、开运算和闭运算 - @@ -69,8 +88,7 @@ Segment-Anything 模型需要手动下载到 - Mask Info: - 显示mask的统计信息,如形状、覆盖率、范围和均值 - MaskHelperLK: - - 如果你忘记了以上`Mask`处理节点的功能,可以使用这个节点查看,因为我知道这个东西只看节点名字很难知道效果 + - 如果你忘记了以上`Mask`处理节点的功能,可以使用这个节点查看,因为我知道以上四个节点只看节点名字很难知道效果 -懒了,今天先写到这,传播民主与自由去了。 diff --git a/example_workflow/detailer_example.json b/example_workflow/detailer_example.json new file mode 100644 index 0000000..d77fb6c --- /dev/null +++ b/example_workflow/detailer_example.json @@ -0,0 +1,997 @@ +{ + "id": "924f8d9b-3083-4698-8424-5d1ff6375ff9", + "revision": 0, + "last_node_id": 23, + "last_link_id": 23, + "nodes": [ + { + "id": 18, + "type": "MaskPreview+", + "pos": [ + 1430, + 580 + ], + "size": [ + 230, + 246 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "localized_name": "mask", + "name": "mask", + "type": "MASK", + "link": 15 + } + ], + "outputs": [], + "title": "BBox result", + "properties": { + "cnr_id": "comfyui_essentials", + "ver": "1.1.0", + "widget_ue_connectable": {}, + "Node name for S&R": "MaskPreview+" + }, + "widgets_values": [] + }, + { + "id": 19, + "type": "MaskPreview+", + "pos": [ + 1670, + 580 + ], + "size": [ + 230, + 246 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "localized_name": "mask", + "name": "mask", + "type": "MASK", + "link": 16 + } + ], + "outputs": [], + "title": "Segmented result", + "properties": { + "cnr_id": "comfyui_essentials", + "ver": "1.1.0", + "widget_ue_connectable": {}, + "Node name for S&R": "MaskPreview+" + }, + "widgets_values": [] + }, + { + "id": 16, + "type": "SAMLoaderLK", + "pos": [ + 1500, + 450 + ], + "size": [ + 261.6646423339844, + 82 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [ + { + "localized_name": "model_name", + "name": "model_name", + "type": "COMBO", + "widget": { + "name": "model_name" + }, + "link": null + }, + { + "localized_name": "device_mode", + "name": "device_mode", + "type": "COMBO", + "widget": { + "name": "device_mode" + }, + "link": null + } + ], + "outputs": [ + { + "localized_name": "SAM_MODEL", + "name": "SAM_MODEL", + "type": "SAM_MODEL", + "links": [ + 13 + ] + } + ], + "properties": { + "aux_id": "LK-168/comfyui_imgutils", + "ver": "4c3019145dd6a17c7c82f4b1d8867b3b8a0077b6", + "widget_ue_connectable": {}, + "Node name for S&R": "SAMLoaderLK" + }, + "widgets_values": [ + "sam_vit_b.pth", + "AUTO" + ] + }, + { + "id": 12, + "type": "ImgutilsGenericDetector", + "pos": [ + 1500, + 200 + ], + "size": [ + 270, + 218 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 10 + }, + { + "localized_name": "detection_type", + "name": "detection_type", + "type": "COMBO", + "widget": { + "name": "detection_type" + }, + "link": null + }, + { + "localized_name": "conf_threshold", + "name": "conf_threshold", + "type": "FLOAT", + "widget": { + "name": "conf_threshold" + }, + "link": null + }, + { + "localized_name": "iou_threshold", + "name": "iou_threshold", + "type": "FLOAT", + "widget": { + "name": "iou_threshold" + }, + "link": null + }, + { + "localized_name": "draw_boxes", + "name": "draw_boxes", + "type": "BOOLEAN", + "widget": { + "name": "draw_boxes" + }, + "link": null + }, + { + "localized_name": "level", + "name": "level", + "type": "STRING", + "widget": { + "name": "level" + }, + "link": null + }, + { + "localized_name": "version", + "name": "version", + "type": "STRING", + "widget": { + "name": "version" + }, + "link": null + } + ], + "outputs": [ + { + "localized_name": "image_with_boxes", + "name": "image_with_boxes", + "type": "IMAGE", + "links": [ + 11 + ] + }, + { + "localized_name": "detection_mask", + "name": "detection_mask", + "type": "MASK", + "links": [ + 15 + ] + }, + { + "localized_name": "detection_bbox", + "name": "detection_bbox", + "type": "BBOX", + "links": [ + 12 + ] + } + ], + "properties": { + "aux_id": "LK-168/comfyui_imgutils", + "ver": "4c3019145dd6a17c7c82f4b1d8867b3b8a0077b6", + "widget_ue_connectable": {}, + "Node name for S&R": "ImgutilsGenericDetector" + }, + "widgets_values": [ + "Hand Detection", + 0.5, + 0.7, + true, + "s", + "" + ] + }, + { + "id": 21, + "type": "MaskToSEGS", + "pos": [ + 2150, + 200 + ], + "size": [ + 270, + 154 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "localized_name": "mask", + "name": "mask", + "type": "MASK", + "link": 18 + }, + { + "localized_name": "combined", + "name": "combined", + "type": "BOOLEAN", + "widget": { + "name": "combined" + }, + "link": null + }, + { + "localized_name": "crop_factor", + "name": "crop_factor", + "type": "FLOAT", + "widget": { + "name": "crop_factor" + }, + "link": null + }, + { + "localized_name": "bbox_fill", + "name": "bbox_fill", + "type": "BOOLEAN", + "widget": { + "name": "bbox_fill" + }, + "link": null + }, + { + "localized_name": "drop_size", + "name": "drop_size", + "type": "INT", + "widget": { + "name": "drop_size" + }, + "link": null + }, + { + "localized_name": "contour_fill", + "name": "contour_fill", + "type": "BOOLEAN", + "widget": { + "name": "contour_fill" + }, + "link": null + } + ], + "outputs": [ + { + "localized_name": "SEGS", + "name": "SEGS", + "type": "SEGS", + "links": [ + 19 + ] + } + ], + "properties": { + "cnr_id": "comfyui-impact-pack", + "ver": "8.17.1", + "widget_ue_connectable": {}, + "Node name for S&R": "MaskToSEGS" + }, + "widgets_values": [ + false, + 3, + false, + 10, + false + ] + }, + { + "id": 13, + "type": "LoadImage", + "pos": [ + 1210, + 200 + ], + "size": [ + 274.080078125, + 314 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "COMBO", + "widget": { + "name": "image" + }, + "link": null + }, + { + "localized_name": "choose file to upload", + "name": "upload", + "type": "IMAGEUPLOAD", + "widget": { + "name": "upload" + }, + "link": null + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 10, + 20, + 21 + ] + }, + { + "localized_name": "MASK", + "name": "MASK", + "type": "MASK", + "links": [] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.41", + "widget_ue_connectable": {}, + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "1067_969609715223178_ST_F_00001_.png", + "image" + ] + }, + { + "id": 14, + "type": "PreviewImage", + "pos": [ + 1920, + 420 + ], + "size": [ + 300, + 420 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "link": 11 + } + ], + "outputs": [], + "title": "Image with boxes", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.41", + "widget_ue_connectable": {}, + "Node name for S&R": "PreviewImage" + }, + "widgets_values": [] + }, + { + "id": 15, + "type": "SAMPredictorNode", + "pos": [ + 1850, + 200 + ], + "size": [ + 270, + 170 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "localized_name": "sam_model", + "name": "sam_model", + "type": "SAM_MODEL", + "link": 13 + }, + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 20 + }, + { + "localized_name": "bbox", + "name": "bbox", + "type": "BBOX", + "link": 12 + }, + { + "localized_name": "threshold", + "name": "threshold", + "type": "FLOAT", + "widget": { + "name": "threshold" + }, + "link": null + }, + { + "localized_name": "points_method", + "name": "points_method", + "type": "COMBO", + "widget": { + "name": "points_method" + }, + "link": null + }, + { + "localized_name": "merge_options", + "name": "merge_options", + "type": "COMBO", + "widget": { + "name": "merge_options" + }, + "link": null + }, + { + "localized_name": "crop_factor", + "name": "crop_factor", + "shape": 7, + "type": "FLOAT", + "widget": { + "name": "crop_factor" + }, + "link": null + } + ], + "outputs": [ + { + "localized_name": "MASK", + "name": "MASK", + "type": "MASK", + "links": [ + 16, + 18, + 23 + ] + }, + { + "localized_name": "SEGS", + "name": "SEGS", + "type": "SEGS", + "links": null + } + ], + "properties": { + "aux_id": "LK-168/comfyui_imgutils", + "ver": "4c3019145dd6a17c7c82f4b1d8867b3b8a0077b6", + "widget_ue_connectable": {}, + "Node name for S&R": "SAMPredictorNode" + }, + "widgets_values": [ + 0.4, + "None", + "Merge All", + 3 + ] + }, + { + "id": 23, + "type": "ImageAndMaskPreview", + "pos": [ + 2240, + 410 + ], + "size": [ + 370, + 430 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "label": "image", + "localized_name": "image", + "name": "image", + "shape": 7, + "type": "IMAGE", + "link": 21 + }, + { + "label": "mask", + "localized_name": "mask", + "name": "mask", + "shape": 7, + "type": "MASK", + "link": 23 + }, + { + "localized_name": "mask_opacity", + "name": "mask_opacity", + "type": "FLOAT", + "widget": { + "name": "mask_opacity" + }, + "link": null + }, + { + "localized_name": "mask_color", + "name": "mask_color", + "type": "STRING", + "widget": { + "name": "mask_color" + }, + "link": null + }, + { + "localized_name": "pass_through", + "name": "pass_through", + "type": "BOOLEAN", + "widget": { + "name": "pass_through" + }, + "link": null + } + ], + "outputs": [ + { + "label": "composite", + "localized_name": "composite", + "name": "composite", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "8950c5fe673f30b7bedee6650ed88e998b6caf27", + "widget_ue_connectable": {}, + "Node name for S&R": "ImageAndMaskPreview" + }, + "widgets_values": [ + 1, + "255, 255, 255", + false + ] + }, + { + "id": 22, + "type": "DetailerForEach", + "pos": [ + 2650, + 190 + ], + "size": [ + 400, + 680 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": null + }, + { + "localized_name": "segs", + "name": "segs", + "type": "SEGS", + "link": 19 + }, + { + "localized_name": "model", + "name": "model", + "type": "MODEL", + "link": null + }, + { + "localized_name": "clip", + "name": "clip", + "type": "CLIP", + "link": null + }, + { + "localized_name": "vae", + "name": "vae", + "type": "VAE", + "link": null + }, + { + "localized_name": "positive", + "name": "positive", + "type": "CONDITIONING", + "link": null + }, + { + "localized_name": "negative", + "name": "negative", + "type": "CONDITIONING", + "link": null + }, + { + "localized_name": "detailer_hook", + "name": "detailer_hook", + "shape": 7, + "type": "DETAILER_HOOK", + "link": null + }, + { + "localized_name": "scheduler_func_opt", + "name": "scheduler_func_opt", + "shape": 7, + "type": "SCHEDULER_FUNC", + "link": null + }, + { + "localized_name": "guide_size", + "name": "guide_size", + "type": "FLOAT", + "widget": { + "name": "guide_size" + }, + "link": null + }, + { + "localized_name": "guide_size_for", + "name": "guide_size_for", + "type": "BOOLEAN", + "widget": { + "name": "guide_size_for" + }, + "link": null + }, + { + "localized_name": "max_size", + "name": "max_size", + "type": "FLOAT", + "widget": { + "name": "max_size" + }, + "link": null + }, + { + "localized_name": "seed", + "name": "seed", + "type": "INT", + "widget": { + "name": "seed" + }, + "link": null + }, + { + "localized_name": "steps", + "name": "steps", + "type": "INT", + "widget": { + "name": "steps" + }, + "link": null + }, + { + "localized_name": "cfg", + "name": "cfg", + "type": "FLOAT", + "widget": { + "name": "cfg" + }, + "link": null + }, + { + "localized_name": "sampler_name", + "name": "sampler_name", + "type": "COMBO", + "widget": { + "name": "sampler_name" + }, + "link": null + }, + { + "localized_name": "scheduler", + "name": "scheduler", + "type": "COMBO", + "widget": { + "name": "scheduler" + }, + "link": null + }, + { + "localized_name": "denoise", + "name": "denoise", + "type": "FLOAT", + "widget": { + "name": "denoise" + }, + "link": null + }, + { + "localized_name": "feather", + "name": "feather", + "type": "INT", + "widget": { + "name": "feather" + }, + "link": null + }, + { + "localized_name": "noise_mask", + "name": "noise_mask", + "type": "BOOLEAN", + "widget": { + "name": "noise_mask" + }, + "link": null + }, + { + "localized_name": "force_inpaint", + "name": "force_inpaint", + "type": "BOOLEAN", + "widget": { + "name": "force_inpaint" + }, + "link": null + }, + { + "localized_name": "wildcard", + "name": "wildcard", + "type": "STRING", + "widget": { + "name": "wildcard" + }, + "link": null + }, + { + "localized_name": "cycle", + "name": "cycle", + "type": "INT", + "widget": { + "name": "cycle" + }, + "link": null + }, + { + "localized_name": "inpaint_model", + "name": "inpaint_model", + "shape": 7, + "type": "BOOLEAN", + "widget": { + "name": "inpaint_model" + }, + "link": null + }, + { + "localized_name": "noise_mask_feather", + "name": "noise_mask_feather", + "shape": 7, + "type": "INT", + "widget": { + "name": "noise_mask_feather" + }, + "link": null + }, + { + "localized_name": "tiled_encode", + "name": "tiled_encode", + "shape": 7, + "type": "BOOLEAN", + "widget": { + "name": "tiled_encode" + }, + "link": null + }, + { + "localized_name": "tiled_decode", + "name": "tiled_decode", + "shape": 7, + "type": "BOOLEAN", + "widget": { + "name": "tiled_decode" + }, + "link": null + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-impact-pack", + "ver": "8.17.1", + "widget_ue_connectable": {}, + "Node name for S&R": "DetailerForEach" + }, + "widgets_values": [ + 512, + true, + 1024, + 121335371194066, + "randomize", + 20, + 8, + "euler", + "normal", + 0.5, + 5, + true, + true, + "", + 1, + false, + 20, + false, + false, + [ + false, + true + ] + ] + } + ], + "links": [ + [ + 10, + 13, + 0, + 12, + 0, + "IMAGE" + ], + [ + 11, + 12, + 0, + 14, + 0, + "IMAGE" + ], + [ + 12, + 12, + 2, + 15, + 2, + "BBOX" + ], + [ + 13, + 16, + 0, + 15, + 0, + "SAM_MODEL" + ], + [ + 15, + 12, + 1, + 18, + 0, + "MASK" + ], + [ + 16, + 15, + 0, + 19, + 0, + "MASK" + ], + [ + 18, + 15, + 0, + 21, + 0, + "MASK" + ], + [ + 19, + 21, + 0, + 22, + 1, + "SEGS" + ], + [ + 20, + 13, + 0, + 15, + 1, + "IMAGE" + ], + [ + 21, + 13, + 0, + 23, + 0, + "IMAGE" + ], + [ + 23, + 15, + 0, + 23, + 1, + "MASK" + ] + ], + "groups": [], + "config": {}, + "extra": { + "workspace_info": { + "id": "M1pq72Nrhan7TltzaAQUJ" + }, + "ue_links": [], + "links_added_by_ue": [], + "ds": { + "scale": 0.7266303355934305, + "offset": [ + -564.8769777196603, + 274.40995013281713 + ] + } + }, + "version": 0.4 +} \ No newline at end of file diff --git a/md_img/bbox_nodes.png b/md_img/bbox_nodes.png new file mode 100644 index 0000000..e864c93 Binary files /dev/null and b/md_img/bbox_nodes.png differ diff --git a/md_img/detailer_example.png b/md_img/detailer_example.png new file mode 100644 index 0000000..998065a Binary files /dev/null and b/md_img/detailer_example.png differ diff --git a/md_img/mask_nodes.png b/md_img/mask_nodes.png new file mode 100644 index 0000000..098a9c5 Binary files /dev/null and b/md_img/mask_nodes.png differ diff --git a/md_img/segment_nodes.png b/md_img/segment_nodes.png new file mode 100644 index 0000000..66f6fc0 Binary files /dev/null and b/md_img/segment_nodes.png differ diff --git a/utils/detect.py b/utils/detect.py index ba888da..d60ad68 100644 --- a/utils/detect.py +++ b/utils/detect.py @@ -108,13 +108,13 @@ class ImgutilsGenericDetector: "draw_boxes": ("BOOLEAN", {"default": True}), # Model specific parameters, always present for simplicity in this generic node "level": ("STRING", {"default": "s", "options": ["n", "s"]}), - "version": ("STRING", {"default": "v1.1", "options": ["v0", "v1", "v1.1"]}), + "version": ("STRING", {"default": "", "options": ["v0", "v1", "v1.1"]}), # "model_name": ("STRING", {"default": ""}), }, } RETURN_TYPES = ("IMAGE", "MASK",BBOX, ) - RETURN_NAMES = ("image_with_boxes", "detection_mask","detection_results",) + RETURN_NAMES = ("image_with_boxes", "detection_mask","detection_bbox",) FUNCTION = "detect" def detect(self, image, detection_type, conf_threshold, diff --git a/utils/mask.py b/utils/mask.py index 2d47a8d..9397502 100644 --- a/utils/mask.py +++ b/utils/mask.py @@ -276,13 +276,17 @@ class MaskInfoNode: } CATEGORY = "imgutils/mask" - RETURN_TYPES = ("MASK", "STRING") - RETURN_NAMES = ("mask", "info") + RETURN_TYPES = ("MASK", "FLOAT", "STRING", "FLOAT", "STRING") + RETURN_NAMES = ("mask", "coverage", "value_range", "mean_value", "detailed_info") FUNCTION = "get_mask_info" def get_mask_info(self, mask): batch_size = mask.shape[0] info_list = [] + total_coverage = 0.0 + total_mean = 0.0 + overall_min = 1.0 + overall_max = 0.0 for i in range(batch_size): current_mask = mask[i].squeeze().cpu().numpy() @@ -297,12 +301,30 @@ class MaskInfoNode: max_val = np.max(current_mask) mean_val = np.mean(current_mask) + # Update overall statistics + total_coverage += coverage + total_mean += mean_val + overall_min = min(overall_min, min_val) + overall_max = max(overall_max, max_val) + info = f"Batch {i}: Shape={shape}, Coverage={coverage:.1f}%, " info += f"Range=[{min_val:.3f}, {max_val:.3f}], Mean={mean_val:.3f}" info_list.append(info) + # Calculate averages + avg_coverage = total_coverage / batch_size + avg_mean = total_mean / batch_size + value_range = overall_max - overall_min + + # Create detailed info combined_info = "\n".join(info_list) - return (mask, combined_info) + if batch_size > 1: + combined_info += f"\n\nOverall Statistics:" + combined_info += f"\nAverage Coverage: {avg_coverage:.1f}%" + combined_info += f"\nOverall Range: [{overall_min:.3f}, {overall_max:.3f}]" + combined_info += f"\nAverage Mean: {avg_mean:.3f}" + + return (mask, avg_coverage, f"[{overall_min:.3f}, {overall_max:.3f}]", avg_mean, combined_info) class MaskHelperLK: """Helper class for above mask operations""" diff --git a/utils/sam.py b/utils/sam.py index ad645b3..880bba6 100644 --- a/utils/sam.py +++ b/utils/sam.py @@ -342,7 +342,7 @@ NODE_CLASS_MAPPINGS = { } NODE_DISPLAY_NAME_MAPPINGS = { - "SAMLoaderLK": "SAM Loader LK", + "SAMLoaderLK": "SAM Loader for SAMPredictorNode", "SAMPredictorNode": "SAM Predictor", } diff --git a/utils/segment.py b/utils/segment.py index 8d27aac..86cb5be 100644 --- a/utils/segment.py +++ b/utils/segment.py @@ -104,7 +104,7 @@ class ImgutilsBBoxSegmenter(object): original_image_pil = Image.fromarray(img_np_255).convert("RGB") bbox_mask_np_0_1 = bbox_mask[0].cpu().numpy() # MASK is [B, H, W], values 0-1 - bbox_mask_np = (bbox_mask_np_0_1 > 0.5).astype(np.float32) + bbox_mask_np = (bbox_mask_np_0_1 > 0.01).astype(np.float32) # If bbox_mask is empty, return original image and empty mask if not np.any(bbox_mask_np): @@ -112,7 +112,17 @@ class ImgutilsBBoxSegmenter(object): empty_mask_tensor = torch.zeros((1, original_image_pil.height, original_image_pil.width), dtype=torch.float32) return (image, empty_mask_tensor) - # Get the main subject's mask from imgutils + # Find bounding box coordinates from the mask + bbox_indices = np.where(bbox_mask_np > 0) + if len(bbox_indices[0]) == 0: + print("BBox mask has no valid pixels. Returning original image and empty mask.") + empty_mask_tensor = torch.zeros((1, original_image_pil.height, original_image_pil.width), dtype=torch.float32) + return (image, empty_mask_tensor) + + min_y, max_y = bbox_indices[0].min(), bbox_indices[0].max() + 1 + min_x, max_x = bbox_indices[1].min(), bbox_indices[1].max() + 1 + + # Get the main subject's mask from imgutils (full image) seg_mask_np_full = get_isnetis_mask(original_image_pil, scale=scale) # Normalize mask to 0-1 float32 if seg_mask_np_full.dtype == np.uint8: @@ -123,17 +133,16 @@ class ImgutilsBBoxSegmenter(object): # Combine imgutils mask with provided bbox_mask final_seg_mask_np = seg_mask_np_full * bbox_mask_np - # Get the segmented image based on mode + # Get the segmented image based on mode (full image) if segment_mode == "rgba_transparent": _, seg_image_pil_full = segment_rgba_with_isnetis(original_image_pil, scale=scale) if seg_image_pil_full.mode != 'RGBA': seg_image_pil_full = seg_image_pil_full.convert('RGBA') # Apply the combined mask to the alpha channel of the segmented image - # Create an alpha channel from the final_seg_mask_np alpha_from_mask = Image.fromarray((final_seg_mask_np * 255).astype(np.uint8), mode='L') seg_image_pil_full.putalpha(alpha_from_mask) - output_image_pil = seg_image_pil_full + full_output_image_pil = seg_image_pil_full else: # RGB modes bg_color_val = 0 if segment_mode == "rgb_black_bg" else 255 @@ -149,20 +158,25 @@ class ImgutilsBBoxSegmenter(object): seg_image_pil_full = seg_image_pil_full.convert('RGB') # Paste the imgutils segmented image onto the background using the combined mask - # Use the final_seg_mask_np to control what gets pasted paste_mask_pil = Image.fromarray((final_seg_mask_np * 255).astype(np.uint8), mode='L') background_image.paste(seg_image_pil_full, (0, 0), paste_mask_pil) - output_image_pil = background_image + full_output_image_pil = background_image + + # Crop the segmented image to the bbox region + cropped_output_image_pil = full_output_image_pil.crop((min_x, min_y, max_x, max_y)) + + # Crop the mask to the bbox region as well + cropped_seg_mask_np = final_seg_mask_np[min_y:max_y, min_x:max_x] # Format output - if output_image_pil: - output_image_np_255 = np.array(output_image_pil.convert("RGB")) + if cropped_output_image_pil: + output_image_np_255 = np.array(cropped_output_image_pil.convert("RGB")) output_image_tensor = torch.from_numpy(output_image_np_255.astype(np.float32) / 255.0).unsqueeze(0) else: # Fallback output_image_tensor = image - # Final mask is the combined mask - output_mask_tensor = torch.from_numpy(final_seg_mask_np).unsqueeze(0) + # Output the cropped mask + output_mask_tensor = torch.from_numpy(cropped_seg_mask_np).unsqueeze(0) return (output_image_tensor, output_mask_tensor)