commit f7061533a1bb17ea1edcd42802acf4baee0a70a8 Author: mengxiangyuan Date: Sat Apr 6 00:24:56 2024 +0800 # 图片转为文字描述(自然语言) diff --git a/ComfyUI-ImageToText.json b/ComfyUI-ImageToText.json new file mode 100644 index 0000000..5447658 --- /dev/null +++ b/ComfyUI-ImageToText.json @@ -0,0 +1,152 @@ +{ + "last_node_id": 4, + "last_link_id": 2, + "nodes": [ + { + "id": 3, + "type": "ComfyUI_ImageToText", + "pos": [ + 726, + 120 + ], + "size": { + "0": 315, + "1": 58 + }, + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 1, + "label": "images" + } + ], + "outputs": [ + { + "name": "text_positive", + "type": "STRING", + "links": [ + 2 + ], + "shape": 3, + "label": "text_positive", + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "ComfyUI_ImageToText" + }, + "widgets_values": [ + "Yes" + ] + }, + { + "id": 2, + "type": "LoadImage", + "pos": [ + 282, + 113 + ], + "size": { + "0": 315, + "1": 314 + }, + "flags": {}, + "order": 0, + "mode": 0, + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1 + ], + "shape": 3, + "label": "图像", + "slot_index": 0 + }, + { + "name": "MASK", + "type": "MASK", + "links": null, + "shape": 3, + "label": "遮罩" + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "00489-MexxL_LCM2_YY-1034089384-1-960-20240111205721.jpg", + "image" + ] + }, + { + "id": 4, + "type": "ShowText|pysssss", + "pos": [ + 1133.3125, + 133.7734375 + ], + "size": [ + 380.328125, + 239.95703125 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "name": "text", + "type": "STRING", + "link": 2, + "widget": { + "name": "text" + }, + "label": "文本" + } + ], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": null, + "shape": 6, + "label": "字符串" + } + ], + "properties": { + "Node name for S&R": "ShowText|pysssss" + }, + "widgets_values": [ + "", + "The image depicts a young girl with black hair seated at a desk, intently focused on her laptop. The desk is adorned with a rainbow-colored record player and a rainbow-colored music note, adding a vibrant touch to the scene. The background is a stark white, providing a contrast that makes the girl and her surroundings stand out. The image is framed by a black border, and a watermark in the bottom right corner reads \"© 2020\"." + ] + } + ], + "links": [ + [ + 1, + 2, + 0, + 3, + 0, + "IMAGE" + ], + [ + 2, + 3, + 0, + 4, + 0, + "STRING" + ] + ], + "groups": [], + "config": {}, + "extra": {}, + "version": 0.4 +} \ No newline at end of file diff --git a/ImageToText.py b/ImageToText.py new file mode 100644 index 0000000..22eafe4 --- /dev/null +++ b/ImageToText.py @@ -0,0 +1,51 @@ +from transformers import AutoModelForCausalLM, AutoTokenizer +from PIL import Image + +import numpy as np + +model_id = "vikhyatk/moondream2" +revision = "2024-04-02" + +class ComfyUI_ImageToText: + def __init__(self): + pass + + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "images": ("IMAGE",), + "log_prompt": (["No", "Yes"], {"default":"Yes"}), + }, + } + + RETURN_TYPES = ('STRING',) + RETURN_NAMES = ('text_positive',) + FUNCTION = "image2text" + OUTPUT_NODE = True + CATEGORY = "ComfyUI_Mexx" + + def image2text(self, images, log_prompt): + pil_images = [] + for image in images: + i = 255. * image.cpu().numpy() + img = Image.fromarray(np.clip(i, 0, 255).astype(np.uint8)) + pil_images.append(img) + image = pil_images[0] + model = AutoModelForCausalLM.from_pretrained( + model_id, trust_remote_code=True, revision=revision + ) + tokenizer = AutoTokenizer.from_pretrained(model_id, revision=revision) + enc_image = model.encode_image(image) + en = model.answer_question(enc_image, "Describe this image.", tokenizer) + if log_prompt == "Yes": + print(f"ImageToText: {en}") + return [en] + +NODE_CLASS_MAPPINGS = { + "ComfyUI_ImageToText": ComfyUI_ImageToText +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "ComfyUI_ImageToText": "ComfyUI_ImageToText" +} diff --git a/README.md b/README.md new file mode 100644 index 0000000..7f2cf88 --- /dev/null +++ b/README.md @@ -0,0 +1,18 @@ +# ComfyUI_ImageToText + +## 功能简述 + +把图片以自然语言描述出来. + +## 使用图例 + +![demo.png](image%2Fdemo.png) + +## 工作流举例 + +SDXL模型下载地址(欢迎点赞点关注): https://www.liblib.art/modelinfo/5913fb0765ce4a4ba210cb1c898df276 +工作流文件(直接拖拽到ComfyUI的页面里即可): [ComfyUI-ImageToText.json](ComfyUI-ImageToText.json) + +## 使用了的模型 + +https://huggingface.co/vikhyatk/moondream2 \ No newline at end of file diff --git a/__init__.py b/__init__.py new file mode 100644 index 0000000..43a80ba --- /dev/null +++ b/__init__.py @@ -0,0 +1,3 @@ +from .ImageToText import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS + +__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS'] diff --git a/image/demo.png b/image/demo.png new file mode 100644 index 0000000..2178f82 Binary files /dev/null and b/image/demo.png differ diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..0df9e5a --- /dev/null +++ b/requirements.txt @@ -0,0 +1,3 @@ +transformers +timm +einops \ No newline at end of file