ollama VLM 支持多个图像输入

This commit is contained in:
qnsh
2025-10-28 14:58:32 +08:00
parent e3525ffd33
commit 5cebd7d735
3 changed files with 26 additions and 18 deletions
+3 -3
View File
@@ -57,9 +57,9 @@
"display_name": "Ollama VLM API",
"description": "This node uses the Ollama VLM model for image reasoning and analysis.",
"inputs": {
"image":{
"name": "image",
"tooltip": "Image used for analysis"
"images":{
"name": "images",
"tooltip": "Images used for analysis"
},
"model": {
"name": "model"
+2 -2
View File
@@ -58,9 +58,9 @@
"display_name": "Ollama 视觉 API",
"description": "这个节点使用Ollama VLM 模型进行图片推理分析",
"inputs": {
"image":{
"images":{
"name": "图像",
"tooltip": "用于分析的图像"
"tooltip": "用于分析的图像(支持多个)"
},
"model": {
"name": "模型"
+21 -13
View File
@@ -66,7 +66,7 @@ class OllamaVLM(io.ComfyNode):
category="YCYY/API/text",
inputs=[
io.Image.Input(
"image",
"images",
tooltip="Image used for analysis"
),
io.String.Input(
@@ -106,7 +106,7 @@ class OllamaVLM(io.ComfyNode):
# return []
# 执行 GeminiImage 节点
@classmethod
def execute(cls,image, system_prompt, user_prompt, model) -> io.NodeOutput:
def execute(cls,images, system_prompt, user_prompt, model) -> io.NodeOutput:
if not user_prompt:
raise ValueError("User prompt cannot be empty")
@@ -123,19 +123,27 @@ class OllamaVLM(io.ComfyNode):
"content": system_prompt
}
payload["messages"].append(system_message)
image_base64 = tensor_to_base64_string(image)
user_message ={
"role": "user",
"content": [
{
"type": "text",
"text": user_prompt
},
{
# 构建用户消息内容,支持多个图片
content = [
{
"type": "text",
"text": user_prompt
}
]
# 处理多个图片
if images is not None:
for image_index in range(images.shape[0]):
image_base64 = tensor_to_base64_string(images[image_index].unsqueeze(0))
content.append({
"type": "image_url",
"image_url": f"data:image/png;base64,{image_base64}"
}
]
})
user_message = {
"role": "user",
"content": content
}
payload["messages"].append(user_message)
try: