From 5cebd7d735a71284c93ba81fc4a0fed8be916a17 Mon Sep 17 00:00:00 2001 From: qnsh Date: Tue, 28 Oct 2025 14:58:32 +0800 Subject: [PATCH] =?UTF-8?q?ollama=20VLM=20=E6=94=AF=E6=8C=81=E5=A4=9A?= =?UTF-8?q?=E4=B8=AA=E5=9B=BE=E5=83=8F=E8=BE=93=E5=85=A5?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- locales/en/nodeDefs.json | 6 +++--- locales/zh/nodeDefs.json | 4 ++-- ollama/ollama_vlm_node.py | 34 +++++++++++++++++++++------------- 3 files changed, 26 insertions(+), 18 deletions(-) diff --git a/locales/en/nodeDefs.json b/locales/en/nodeDefs.json index 582ee96..82adb64 100644 --- a/locales/en/nodeDefs.json +++ b/locales/en/nodeDefs.json @@ -57,9 +57,9 @@ "display_name": "Ollama VLM API", "description": "This node uses the Ollama VLM model for image reasoning and analysis.", "inputs": { - "image":{ - "name": "image", - "tooltip": "Image used for analysis" + "images":{ + "name": "images", + "tooltip": "Images used for analysis" }, "model": { "name": "model" diff --git a/locales/zh/nodeDefs.json b/locales/zh/nodeDefs.json index 4f91bbe..5de2a05 100644 --- a/locales/zh/nodeDefs.json +++ b/locales/zh/nodeDefs.json @@ -58,9 +58,9 @@ "display_name": "Ollama 视觉 API", "description": "这个节点使用Ollama VLM 模型进行图片推理分析", "inputs": { - "image":{ + "images":{ "name": "图像", - "tooltip": "用于分析的图像" + "tooltip": "用于分析的图像(支持多个)" }, "model": { "name": "模型" diff --git a/ollama/ollama_vlm_node.py b/ollama/ollama_vlm_node.py index 81c6d8f..628cdcd 100644 --- a/ollama/ollama_vlm_node.py +++ b/ollama/ollama_vlm_node.py @@ -66,7 +66,7 @@ class OllamaVLM(io.ComfyNode): category="YCYY/API/text", inputs=[ io.Image.Input( - "image", + "images", tooltip="Image used for analysis" ), io.String.Input( @@ -106,7 +106,7 @@ class OllamaVLM(io.ComfyNode): # return [] # 执行 GeminiImage 节点 @classmethod - def execute(cls,image, system_prompt, user_prompt, model) -> io.NodeOutput: + def execute(cls,images, system_prompt, user_prompt, model) -> io.NodeOutput: if not user_prompt: raise ValueError("User prompt cannot be empty") @@ -123,19 +123,27 @@ class OllamaVLM(io.ComfyNode): "content": system_prompt } payload["messages"].append(system_message) - image_base64 = tensor_to_base64_string(image) - user_message ={ - "role": "user", - "content": [ - { - "type": "text", - "text": user_prompt - }, - { + + # 构建用户消息内容,支持多个图片 + content = [ + { + "type": "text", + "text": user_prompt + } + ] + + # 处理多个图片 + if images is not None: + for image_index in range(images.shape[0]): + image_base64 = tensor_to_base64_string(images[image_index].unsqueeze(0)) + content.append({ "type": "image_url", "image_url": f"data:image/png;base64,{image_base64}" - } - ] + }) + + user_message = { + "role": "user", + "content": content } payload["messages"].append(user_message) try: