diff --git a/Example/ImageToPromptExample.json b/Example/ImageToPromptExample.json new file mode 100644 index 0000000..e2a4086 --- /dev/null +++ b/Example/ImageToPromptExample.json @@ -0,0 +1,765 @@ +{ + "id": "9ae6082b-c7f4-433c-9971-7a8f65a3ea65", + "revision": 0, + "last_node_id": 50, + "last_link_id": 48, + "nodes": [ + { + "id": 39, + "type": "CLIPLoader", + "pos": [ + 130.2638101844517, + 435.2545050421948 + ], + "size": [ + 270, + 106 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 44 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.73", + "Node name for S&R": "CLIPLoader", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "models": [ + { + "name": "qwen_3_4b.safetensors", + "url": "https://huggingface.co/Comfy-Org/z_image_turbo/resolve/main/split_files/text_encoders/qwen_3_4b.safetensors", + "directory": "text_encoders" + } + ], + "widget_ue_connectable": {} + }, + "widgets_values": [ + "qwen_3_4b.safetensors", + "lumina2", + "default" + ] + }, + { + "id": 40, + "type": "VAELoader", + "pos": [ + 130.2638101844517, + 585.2545050421948 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "VAE", + "type": "VAE", + "links": [ + 39 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.73", + "Node name for S&R": "VAELoader", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "models": [ + { + "name": "ae.safetensors", + "url": "https://huggingface.co/Comfy-Org/z_image_turbo/resolve/main/split_files/vae/ae.safetensors", + "directory": "vae" + } + ], + "widget_ue_connectable": {} + }, + "widgets_values": [ + "ae.safetensors" + ] + }, + { + "id": 42, + "type": "ConditioningZeroOut", + "pos": [ + 660.2638101844517, + 725.2545050421948 + ], + "size": [ + 197.712890625, + 26 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "conditioning", + "type": "CONDITIONING", + "link": 36 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 42 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.73", + "Node name for S&R": "ConditioningZeroOut", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [] + }, + { + "id": 46, + "type": "UNETLoader", + "pos": [ + 130.2638101844517, + 305.2545050421948 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 37 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.73", + "Node name for S&R": "UNETLoader", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "models": [ + { + "name": "z_image_turbo_bf16.safetensors", + "url": "https://huggingface.co/Comfy-Org/z_image_turbo/resolve/main/split_files/diffusion_models/z_image_turbo_bf16.safetensors", + "directory": "diffusion_models" + } + ], + "widget_ue_connectable": {} + }, + "widgets_values": [ + "z_image_turbo_bf16.safetensors", + "default" + ] + }, + { + "id": 44, + "type": "KSampler", + "pos": [ + 900.2638101844517, + 375.2545050421948 + ], + "size": [ + 315, + 474 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 40 + }, + { + "name": "positive", + "type": "CONDITIONING", + "link": 41 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 42 + }, + { + "name": "latent_image", + "type": "LATENT", + "link": 43 + } + ], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "slot_index": 0, + "links": [ + 38 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.64", + "Node name for S&R": "KSampler", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [ + 1071032685239444, + "randomize", + 9, + 1, + "res_multistep", + "simple", + 1 + ] + }, + { + "id": 43, + "type": "VAEDecode", + "pos": [ + 1240, + 170 + ], + "size": [ + 210, + 46 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 38 + }, + { + "name": "vae", + "type": "VAE", + "link": 39 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "slot_index": 0, + "links": [ + 45 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.64", + "Node name for S&R": "VAEDecode", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [] + }, + { + "id": 47, + "type": "ModelSamplingAuraFlow", + "pos": [ + 900.2638101844517, + 265.2545050421948 + ], + "size": [ + 310, + 60 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 37 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "slot_index": 0, + "links": [ + 40 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.64", + "Node name for S&R": "ModelSamplingAuraFlow", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [ + 3 + ] + }, + { + "id": 45, + "type": "CLIPTextEncode", + "pos": [ + 450.2638101844517, + 305.2545050421948 + ], + "size": [ + 410, + 370 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 44 + }, + { + "name": "text", + "type": "STRING", + "widget": { + "name": "text" + }, + "link": 46 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 36, + 41 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.73", + "Node name for S&R": "CLIPTextEncode", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Latina female with thick wavy hair, harbor boats and pastel houses behind. Breezy seaside light, warm tones, cinematic close-up." + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 49, + "type": "LoadImage", + "pos": [ + -620, + 280 + ], + "size": [ + 260, + 500 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 47 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.76", + "widget_ue_connectable": {}, + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "initial_image (1).png", + "image" + ] + }, + { + "id": 41, + "type": "EmptySD3LatentImage", + "pos": [ + 130.2638101844517, + 735.2545050421948 + ], + "size": [ + 260, + 110 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "slot_index": 0, + "links": [ + 43 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.64", + "Node name for S&R": "EmptySD3LatentImage", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [ + 720, + 1280, + 1 + ] + }, + { + "id": 48, + "type": "GGUFInference", + "pos": [ + -320, + 300 + ], + "size": [ + 400, + 426 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "image", + "shape": 7, + "type": "IMAGE", + "link": 47 + } + ], + "outputs": [ + { + "name": "text", + "type": "STRING", + "links": [ + 46 + ] + }, + { + "name": "used_seed", + "type": "INT", + "links": null + } + ], + "properties": { + "cnr_id": "Listhelper", + "ver": "11beb2ddf5584d1b02a51de1d1483b97b5979b0f", + "widget_ue_connectable": {}, + "Node name for S&R": "GGUFInference" + }, + "widgets_values": [ + "Huihui-Qwen3-VL-4B-Instruct-abliterated-Q4_K_M.gguf", + "描述這張圖片", + "image_to_prompt.md", + "", + 4096, + 0.7, + 0.9, + 40, + 667688996286694, + "randomize", + false, + "mmproj-F16.gguf", + false + ] + }, + { + "id": 9, + "type": "SaveImage", + "pos": [ + 1240, + 260 + ], + "size": [ + 390, + 660 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 45 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.64", + "Node name for S&R": "SaveImage", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [ + "z-image" + ] + } + ], + "links": [ + [ + 36, + 45, + 0, + 42, + 0, + "CONDITIONING" + ], + [ + 37, + 46, + 0, + 47, + 0, + "MODEL" + ], + [ + 38, + 44, + 0, + 43, + 0, + "LATENT" + ], + [ + 39, + 40, + 0, + 43, + 1, + "VAE" + ], + [ + 40, + 47, + 0, + 44, + 0, + "MODEL" + ], + [ + 41, + 45, + 0, + 44, + 1, + "CONDITIONING" + ], + [ + 42, + 42, + 0, + 44, + 2, + "CONDITIONING" + ], + [ + 43, + 41, + 0, + 44, + 3, + "LATENT" + ], + [ + 44, + 39, + 0, + 45, + 0, + "CLIP" + ], + [ + 45, + 43, + 0, + 9, + 0, + "IMAGE" + ], + [ + 46, + 48, + 0, + 45, + 1, + "STRING" + ], + [ + 47, + 49, + 0, + 48, + 0, + "IMAGE" + ] + ], + "groups": [ + { + "id": 2, + "title": "Step2 - Image size", + "bounding": [ + 120, + 670, + 290, + 200 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 3, + "title": "Step3 - Prompt", + "bounding": [ + 430, + 240, + 450, + 540 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 4, + "title": "Step1 - Load models", + "bounding": [ + 120, + 240, + 290, + 413.6 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], + "config": {}, + "extra": { + "ds": { + "scale": 0.9243930692394946, + "offset": [ + 720, + -70 + ] + }, + "frontendVersion": "1.33.10", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true, + "workflowRendererVersion": "LG", + "ue_links": [], + "links_added_by_ue": [] + }, + "version": 0.4 +} \ No newline at end of file diff --git a/Prompt/image_edit.md b/Prompt/image_edit.md new file mode 100644 index 0000000..9e15a05 --- /dev/null +++ b/Prompt/image_edit.md @@ -0,0 +1,124 @@ +# Image Analysis & Element Replacement Editor + +You are an advanced image analysis and prompt editing AI. Your task is to: +1. Analyze the provided image and generate a comprehensive English prompt +2. Understand the user's edit instruction +3. Intelligently modify the prompt to reflect the requested changes + +## Core Workflow: + +### Step 1: Image Analysis +Analyze the image thoroughly and create a detailed English prompt that captures: +- **Main subjects**: People, objects, animals +- **Appearance details**: Colors, textures, materials, styles +- **Actions & poses**: What subjects are doing +- **Environment**: Setting, background, location +- **Composition**: Layout, perspective, framing +- **Lighting**: Direction, quality, mood +- **Style**: Art style, photography type +- **Quality markers**: Resolution, detail level + +### Step 2: Understanding Edit Instructions +The user will provide editing instructions in natural language (English or Chinese), such as: +- "Change the dress to pink" / "把衣服換成粉紅色" +- "Make it nighttime" / "改成夜晚場景" +- "Add a cat in the background" / "背景加入一隻貓" +- "Change the hairstyle to short" / "把髮型改成短髮" +- "Replace the car with a bicycle" / "把車子換成腳踏車" + +### Step 3: Intelligent Prompt Modification +- **Identify** the specific element in the original prompt that corresponds to the user's instruction +- **Replace** that element with the new description while maintaining coherence +- **Preserve** all other elements that weren't mentioned in the edit instruction +- **Ensure** the modified prompt remains natural and grammatically correct +- **Maintain** the style and quality descriptors unless specifically changed + +## Element Replacement Rules: + +1. **Clothing Changes**: Replace only the clothing description, keep the pose, person characteristics, and other details + - Original: "woman wearing blue dress" + - Instruction: "change to red jacket" + - Modified: "woman wearing red jacket" + +2. **Color Changes**: Replace the specific color attribute + - Original: "red sports car" + - Instruction: "make it silver" + - Modified: "silver sports car" + +3. **Object Replacement**: Replace the entire object while maintaining its role in the composition + - Original: "man holding coffee cup" + - Instruction: "change coffee to wine glass" + - Modified: "man holding wine glass" + +4. **Environmental Changes**: Modify setting while keeping subjects intact + - Original: "beach at sunset" + - Instruction: "change to mountain landscape" + - Modified: "mountain landscape at sunset" + +5. **Addition Requests**: Integrate new elements naturally + - Original: "empty room with white walls" + - Instruction: "add a painting on the wall" + - Modified: "room with white walls and a framed painting" + +6. **Style Changes**: Replace style descriptors globally + - Original: "photorealistic portrait" + - Instruction: "make it anime style" + - Modified: "anime style portrait" + +## Output Format: + +**CRITICAL INSTRUCTION**: Your response must ONLY contain the final modified English prompt. Do NOT include: +- ❌ Any explanations or commentary (in Chinese or English) +- ❌ Phrases like "這是...", "以下是...", "根據您的要求..." +- ❌ Section headers like "Original Prompt:", "Modified Prompt:", etc. +- ❌ Analysis or reasoning about changes +- ❌ Any introductory or concluding statements + +**OUTPUT REQUIREMENTS**: +- ✅ Output ONLY the final modified English prompt +- ✅ The prompt should be ready for direct use in image generation +- ✅ Single paragraph format with comma-separated elements +- ✅ Professional, natural English language + +## Important Guidelines: + +- **Be precise**: Only change what the user explicitly asks to change +- **Maintain coherence**: Ensure the modified prompt makes logical sense +- **Preserve quality**: Keep technical quality descriptors (4k, detailed, etc.) +- **Natural language**: The output should read as a natural, cohesive prompt +- **Element relationships**: Consider how changes affect related elements +- **Cultural understanding**: Understand editing instructions in both English and Chinese +- **Context awareness**: Some changes may require adjusting multiple related elements +- **Direct output only**: NO explanations, NO comments, ONLY the modified prompt + +## Examples: + +### Example 1: +**Input Image**: [Photo of a young woman in blue dress standing in a garden] +**User Instruction**: "把衣服換成粉紅色洋裝" + +**CORRECT Output** (only this): +``` +A young woman with long brown hair, gentle smile, wearing elegant pink dress, standing in a sunlit garden, soft natural lighting, warm color palette, professional portrait photography, high resolution +``` + +**WRONG Output** (DO NOT do this): +``` +這是一個非常清楚的編輯請求,我將根據您的指示將女孩的衣物改為粉色洋裝。 +以下是符合您要求的修改後提示: + +A young woman with long brown hair, gentle smile, wearing elegant pink dress, standing in a sunlit garden, soft natural lighting, warm color palette, professional portrait photography, high resolution +``` + +### Example 2: +**Input Image**: [Beach scene with red car] +**User Instruction**: "change the car to blue" + +**CORRECT Output** (only this): +``` +Blue sports car parked on sandy beach, ocean waves in background, sunset lighting, golden hour, professional automotive photography, high resolution, detailed +``` + +--- + +Now, analyze the provided image and wait for the user's editing instruction. When you receive the instruction, output ONLY the modified English prompt with NO additional text. diff --git a/Prompt/image_to_prompt.md b/Prompt/image_to_prompt.md new file mode 100644 index 0000000..22b83a6 --- /dev/null +++ b/Prompt/image_to_prompt.md @@ -0,0 +1,61 @@ +# Image to Prompt Analyzer + +You are a professional image analysis expert specialized in converting images into detailed, high-quality English prompts for image generation. Your task is to analyze the provided image and generate a comprehensive English prompt that can be used to recreate a similar or nearly identical image. + +## Analysis Guidelines: + +1. **Subject Analysis**: Identify the main subject(s) in the image + - People: age, gender, facial features, expression, pose, clothing + - Objects: type, color, material, condition, placement + - Animals: species, characteristics, actions, appearance + +2. **Composition & Layout**: Describe the visual arrangement + - Camera angle and perspective + - Subject positioning and framing + - Rule of thirds, symmetry, or other compositional techniques + +3. **Style & Aesthetic**: Identify the artistic style + - Photography style (portrait, landscape, macro, etc.) + - Art style (realistic, anime, oil painting, watercolor, etc.) + - Overall mood and atmosphere + +4. **Colors & Lighting**: Analyze the color palette and lighting + - Dominant colors and color harmony + - Light source, direction, and quality + - Shadows, highlights, and contrast + - Time of day and lighting mood + +5. **Technical Details**: Note specific technical aspects + - Depth of field (bokeh, blur) + - Image quality markers (high resolution, sharp, detailed) + - Special effects or post-processing + +6. **Background & Environment**: Describe the setting + - Location type and characteristics + - Background elements and details + - Spatial relationship between foreground and background + +## Output Format: + +Generate a single, comprehensive English prompt following this structure: + +``` +[Main Subject], [Key Characteristics], [Action/Pose], [Clothing/Appearance Details], [Setting/Environment], [Composition], [Lighting], [Color Palette], [Style], [Technical Quality], [Additional Details] +``` + +**Important Guidelines:** +- Use comma-separated phrases for clarity +- Start with the most important elements +- Be specific but concise +- Use professional photography and art terminology +- Include quality markers like "high quality", "detailed", "8k resolution", "professional photography" +- Focus on visual elements that can be recreated +- Avoid subjective interpretations +- Use descriptive adjectives effectively + +## Example Output: + +For a portrait photo, your output might be: +"A young woman with long flowing brown hair, gentle smile, wearing elegant white dress, standing in a sunlit garden, soft natural lighting from the left, warm color palette with golden tones, shallow depth of field, blurred background with green foliage, professional portrait photography, high resolution, detailed skin texture, cinematic composition" + +Now, analyze the provided image and generate a detailed English prompt that captures all essential visual elements for image recreation. diff --git a/gguf_inference.py b/gguf_inference.py index aeeea3e..f44b026 100644 --- a/gguf_inference.py +++ b/gguf_inference.py @@ -19,10 +19,12 @@ SUGGESTED_MODELS = { "Download: Z-Image (Abliterated)": "https://huggingface.co/Mungert/Qwen3-4B-abliterated-GGUF/resolve/main/Qwen3-4B-abliterated-q4_k_m.gguf", "Download: Qwen": "https://huggingface.co/unsloth/Qwen2.5-VL-7B-Instruct-GGUF/resolve/main/Qwen2.5-VL-7B-Instruct-Q4_K_M.gguf", "Download: Qwen (Abliterated)": "https://huggingface.co/mradermacher/Qwen2.5-VL-7B-Instruct-abliterated-GGUF/resolve/main/Qwen2.5-VL-7B-Instruct-abliterated.Q4_K_M.gguf", + "Download: QwenVL": "https://huggingface.co/mradermacher/Huihui-Qwen3-VL-4B-Instruct-abliterated-GGUF/resolve/main/Huihui-Qwen3-VL-4B-Instruct-abliterated.Q4_K_M.gguf", } SUGGESTED_MMPROJ = { "Download: mmproj": "https://huggingface.co/unsloth/Qwen2.5-VL-7B-Instruct-GGUF/resolve/main/mmproj-F16.gguf", + "Download: QwenVL mmproj": "https://huggingface.co/noctrex/Huihui-Qwen3-VL-4B-Instruct-abliterated-i1-GGUF/resolve/main/mmproj-F16.gguf", } class GGUFInference: @@ -35,6 +37,7 @@ class GGUFInference: def __init__(self): self.model = None self.current_model_path = None + self.current_mmproj_path = None self.clip_model_array = None self.llama_cpp_available = False self._check_llama_cpp() @@ -250,16 +253,12 @@ class GGUFInference: "default": False, "tooltip": "Keep model in memory after inference" }), - "enable_vision": ("BOOLEAN", { - "default": False, - "tooltip": "Enable vision model (requires mmproj file)" - }), "mmproj_file": (mmproj_names, { "default": mmproj_names[0] if mmproj_names else "No mmproj files", - "tooltip": "Vision model mmproj file" + "tooltip": "Vision model mmproj file (auto-enabled when image is provided and model is VL type)" }), "image": ("IMAGE", { - "tooltip": "Input image for vision model" + "tooltip": "Input image for vision model (auto-enables vision mode for VL models)" }), "auto_install_llama_cpp": ("BOOLEAN", { "default": False, @@ -281,6 +280,9 @@ class GGUFInference: del self.model self.model = None + self.current_model_path = None + self.current_mmproj_path = None + if self.clip_model_array is not None: del self.clip_model_array self.clip_model_array = None @@ -359,14 +361,21 @@ class GGUFInference: def _load_model(self, model_path: str, enable_vision: bool = False, mmproj_path: Optional[str] = None) -> bool: """Load GGUF model with llama-cpp-python""" try: - # Check if model is already loaded - if self.model is not None and self.current_model_path == model_path: + # Check if model and mmproj are already loaded + if (self.model is not None and + self.current_model_path == model_path and + self.current_mmproj_path == mmproj_path): print(f"Model already loaded: {os.path.basename(model_path)}") + if mmproj_path: + print(f" with mmproj: {os.path.basename(mmproj_path)}") return True - # Unload previous model + # Unload previous model if model or mmproj changed if self.model is not None: - print("Unloading previous model...") + if self.current_model_path != model_path: + print("Unloading previous model (model changed)...") + elif self.current_mmproj_path != mmproj_path: + print("Unloading previous model (mmproj changed)...") self._free_memory() if not self.llama_cpp_available: @@ -412,6 +421,7 @@ class GGUFInference: self.model = Llama(**load_kwargs) self.current_model_path = model_path + self.current_mmproj_path = mmproj_path load_time = time.time() - load_start print(f"Model loaded successfully (Time: {load_time:.2f}s)") @@ -423,6 +433,7 @@ class GGUFInference: traceback.print_exc() self.model = None self.current_model_path = None + self.current_mmproj_path = None return False def _remove_thinking_tags(self, text: str) -> str: @@ -475,7 +486,6 @@ class GGUFInference: top_k: int, seed: int = 0, keep_model_loaded: bool = False, - enable_vision: bool = False, mmproj_file: str = "No mmproj files", image = None, auto_install_llama_cpp: bool = False, @@ -547,20 +557,28 @@ class GGUFInference: # Check if this is a vision model is_vision_model = self._is_vision_model(model_path) - # Get mmproj path if vision is enabled and model supports it + # Auto-detect vision mode: enable if image is provided and model is VL type + enable_vision = False + if image is not None and is_vision_model: + enable_vision = True + print("=" * 70) + print("Auto-detected: Vision mode enabled") + print(f" - Image input: Provided") + print(f" - Model type: VL (Vision-Language)") + print("=" * 70) + elif image is not None and not is_vision_model: + print("=" * 70) + print("WARNING: Image provided but model is not a VL (Vision-Language) type.") + print(f"Model: {os.path.basename(model_path)}") + print("Vision mode will NOT be enabled. Image will be ignored.") + print("=" * 70) + + # Get mmproj path if vision is enabled mmproj_path = None if enable_vision: - if not is_vision_model: + if mmproj_file == "No mmproj files": print("=" * 70) - print("WARNING: Vision mode is enabled but model is not a vision model (VL).") - print(f"Model: {os.path.basename(model_path)}") - print("Ignoring vision mode and mmproj settings.") - print("Processing as text-only model.") - print("=" * 70) - enable_vision = False - elif mmproj_file == "No mmproj files": - print("=" * 70) - print("WARNING: Vision model detected but no mmproj file selected.") + print("WARNING: Vision mode detected but no mmproj file selected.") print("Falling back to text-only mode.") print("=" * 70) enable_vision = False @@ -654,7 +672,7 @@ class GGUFInference: "role": "user", "content": [ {"type": "text", "text": prompt}, - {"type": "image_url", "image_url": {"url": image_url}} + {"type": "image_url", "image_url": image_url} ] }) print("Using vision mode with image input")