From 75abbbf4f246d8110e2e76b39bc4cc50ad060377 Mon Sep 17 00:00:00 2001 From: dseditor Date: Thu, 4 Dec 2025 17:03:21 +0800 Subject: [PATCH] zimage --- Example/ZimageDefault.json | 751 +++++++++++++++++++++++++++++++++++++ README.md | 128 ++++--- 2 files changed, 819 insertions(+), 60 deletions(-) create mode 100644 Example/ZimageDefault.json diff --git a/Example/ZimageDefault.json b/Example/ZimageDefault.json new file mode 100644 index 0000000..dce4da7 --- /dev/null +++ b/Example/ZimageDefault.json @@ -0,0 +1,751 @@ +{ + "id": "9ae6082b-c7f4-433c-9971-7a8f65a3ea65", + "revision": 0, + "last_node_id": 49, + "last_link_id": 48, + "nodes": [ + { + "id": 39, + "type": "CLIPLoader", + "pos": [ + 130.2638101844517, + 435.2545050421948 + ], + "size": [ + 270, + 106 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 44 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.73", + "Node name for S&R": "CLIPLoader", + "models": [ + { + "name": "qwen_3_4b.safetensors", + "url": "https://huggingface.co/Comfy-Org/z_image_turbo/resolve/main/split_files/text_encoders/qwen_3_4b.safetensors", + "directory": "text_encoders" + } + ], + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [ + "qwen_3_4b.safetensors", + "lumina2", + "default" + ] + }, + { + "id": 40, + "type": "VAELoader", + "pos": [ + 130.2638101844517, + 585.2545050421948 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "VAE", + "type": "VAE", + "links": [ + 39 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.73", + "Node name for S&R": "VAELoader", + "models": [ + { + "name": "ae.safetensors", + "url": "https://huggingface.co/Comfy-Org/z_image_turbo/resolve/main/split_files/vae/ae.safetensors", + "directory": "vae" + } + ], + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [ + "ae.safetensors" + ] + }, + { + "id": 42, + "type": "ConditioningZeroOut", + "pos": [ + 660.2638101844517, + 725.2545050421948 + ], + "size": [ + 197.712890625, + 26 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "conditioning", + "type": "CONDITIONING", + "link": 36 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 42 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.73", + "Node name for S&R": "ConditioningZeroOut", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [] + }, + { + "id": 41, + "type": "EmptySD3LatentImage", + "pos": [ + 130.2638101844517, + 735.2545050421948 + ], + "size": [ + 260, + 110 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "slot_index": 0, + "links": [ + 43 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.64", + "Node name for S&R": "EmptySD3LatentImage", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [ + 1024, + 1024, + 1 + ] + }, + { + "id": 9, + "type": "SaveImage", + "pos": [ + 1240, + 260 + ], + "size": [ + 780, + 660 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 45 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.64", + "Node name for S&R": "SaveImage", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [ + "z-image" + ] + }, + { + "id": 44, + "type": "KSampler", + "pos": [ + 900.2638101844517, + 375.2545050421948 + ], + "size": [ + 315, + 474 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 40 + }, + { + "name": "positive", + "type": "CONDITIONING", + "link": 41 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 42 + }, + { + "name": "latent_image", + "type": "LATENT", + "link": 43 + } + ], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "slot_index": 0, + "links": [ + 38 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.64", + "Node name for S&R": "KSampler", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [ + 410513707389275, + "randomize", + 9, + 1, + "res_multistep", + "simple", + 1 + ] + }, + { + "id": 43, + "type": "VAEDecode", + "pos": [ + 1240, + 170 + ], + "size": [ + 210, + 46 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 38 + }, + { + "name": "vae", + "type": "VAE", + "link": 39 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "slot_index": 0, + "links": [ + 45 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.64", + "Node name for S&R": "VAEDecode", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [] + }, + { + "id": 47, + "type": "ModelSamplingAuraFlow", + "pos": [ + 900.2638101844517, + 265.2545050421948 + ], + "size": [ + 310, + 60 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 37 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "slot_index": 0, + "links": [ + 40 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.64", + "Node name for S&R": "ModelSamplingAuraFlow", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [ + 3 + ] + }, + { + "id": 35, + "type": "MarkdownNote", + "pos": [ + -390, + 270 + ], + "size": [ + 490, + 400 + ], + "flags": { + "collapsed": false + }, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [], + "title": "Model link", + "properties": {}, + "widgets_values": [ + "## Report issue\n\nIf you found any issues when running this workflow, [report template issue here](https://github.com/Comfy-Org/workflow_templates/issues)\n\n\n## Model links\n\n**text_encoders**\n\n- [qwen_3_4b.safetensors](https://huggingface.co/Comfy-Org/z_image_turbo/resolve/main/split_files/text_encoders/qwen_3_4b.safetensors)\n\n**diffusion_models**\n\n- [z_image_turbo_bf16.safetensors](https://huggingface.co/Comfy-Org/z_image_turbo/resolve/main/split_files/diffusion_models/z_image_turbo_bf16.safetensors)\n\n**vae**\n\n- [ae.safetensors](https://huggingface.co/Comfy-Org/z_image_turbo/resolve/main/split_files/vae/ae.safetensors)\n\n\nModel Storage Location\n\n```\n📂 ComfyUI/\n├── 📂 models/\n│ ├── 📂 text_encoders/\n│ │ └── qwen_3_4b.safetensors\n│ ├── 📂 diffusion_models/\n│ │ └── z_image_turbo_bf16.safetensors\n│ └── 📂 vae/\n│ └── ae.safetensors\n```\n\n" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 45, + "type": "CLIPTextEncode", + "pos": [ + 450.2638101844517, + 305.2545050421948 + ], + "size": [ + 410, + 370 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 44 + }, + { + "name": "text", + "type": "STRING", + "widget": { + "name": "text" + }, + "link": 47 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 36, + 41 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.73", + "Node name for S&R": "CLIPTextEncode", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [ + "Latina female with thick wavy hair, harbor boats and pastel houses behind. Breezy seaside light, warm tones, cinematic close-up." + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 46, + "type": "UNETLoader", + "pos": [ + 130.2638101844517, + 305.2545050421948 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 37 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.73", + "Node name for S&R": "UNETLoader", + "models": [ + { + "name": "z_image_turbo_bf16.safetensors", + "url": "https://huggingface.co/Comfy-Org/z_image_turbo/resolve/main/split_files/diffusion_models/z_image_turbo_bf16.safetensors", + "directory": "diffusion_models" + } + ], + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [ + "z_image_turbo_bf16.safetensors", + "default" + ] + }, + { + "id": 48, + "type": "QwenGPUInference", + "pos": [ + 137.4698090330546, + -69.64821666321234 + ], + "size": [ + 400, + 286 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "text", + "type": "STRING", + "links": [ + 47, + 48 + ] + } + ], + "properties": { + "cnr_id": "Listhelper", + "ver": "0599db202ae2b6f1b283bd90b03420ea2fa541e3", + "Node name for S&R": "QwenGPUInference" + }, + "widgets_values": [ + "Latina female with thick wavy hair, harbor boats and pastel houses behind. Breezy seaside light, warm tones, cinematic close-up.", + "photography_en.md", + "", + 2048, + 0.7, + true, + 0.9, + 50 + ] + }, + { + "id": 49, + "type": "PreviewAny", + "pos": [ + 584.1183999841683, + -68.82863434298851 + ], + "size": [ + 140, + 76 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "source", + "type": "*", + "link": 48 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.76", + "Node name for S&R": "PreviewAny" + }, + "widgets_values": [] + } + ], + "links": [ + [ + 36, + 45, + 0, + 42, + 0, + "CONDITIONING" + ], + [ + 37, + 46, + 0, + 47, + 0, + "MODEL" + ], + [ + 38, + 44, + 0, + 43, + 0, + "LATENT" + ], + [ + 39, + 40, + 0, + 43, + 1, + "VAE" + ], + [ + 40, + 47, + 0, + 44, + 0, + "MODEL" + ], + [ + 41, + 45, + 0, + 44, + 1, + "CONDITIONING" + ], + [ + 42, + 42, + 0, + 44, + 2, + "CONDITIONING" + ], + [ + 43, + 41, + 0, + 44, + 3, + "LATENT" + ], + [ + 44, + 39, + 0, + 45, + 0, + "CLIP" + ], + [ + 45, + 43, + 0, + 9, + 0, + "IMAGE" + ], + [ + 47, + 48, + 0, + 45, + 1, + "STRING" + ], + [ + 48, + 48, + 0, + 49, + 0, + "*" + ] + ], + "groups": [ + { + "id": 2, + "title": "Step2 - Image size", + "bounding": [ + 120.2638101844517, + 665.2545050421948, + 290, + 200 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 3, + "title": "Step3 - Prompt", + "bounding": [ + 430.2638101844517, + 235.2545050421948, + 450, + 540 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 4, + "title": "Step1 - Load models", + "bounding": [ + 120.2638101844517, + 235.2545050421948, + 290, + 413.6 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], + "config": {}, + "extra": { + "ds": { + "scale": 1.0168323761634441, + "offset": [ + 24.38891495737217, + 115.54233181515173 + ] + }, + "frontendVersion": "1.32.10", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true, + "workflowRendererVersion": "LG" + }, + "version": 0.4 +} \ No newline at end of file diff --git a/README.md b/README.md index db011b5..1bf7ed3 100644 --- a/README.md +++ b/README.md @@ -255,65 +255,53 @@ Output: ["Chapter3", "Chapter1"] (randomized) ### Overview -The **QwenGPUInference** node is an AI-powered photo prompt optimizer that transforms simple scene descriptions into detailed, professional photography prompts. It uses the Qwen3-4B language model with intelligent GPU memory management and CPU offload support for seamless integration with ComfyUI. +The **QwenGPUInference** node is an AI-powered text generation node that uses the Qwen3-4B language model with intelligent GPU memory management and CPU offload support for seamless integration with ComfyUI. It can transform simple descriptions into detailed, professional prompts using customizable templates. ### Features -- **Automatic Model Detection**: Finds qwen_3_4b.safetensors in text_encoders folder +- **Automatic Model Detection**: Automatically finds Qwen safetensors files in `text_encoders` or `clip` folders - **Smart Memory Management**: Automatically detects GPU memory and chooses optimal loading strategy -- **CPU Offload Support**: Works alongside other ComfyUI models without memory conflicts -- **Professional Prompt Generation**: Transforms simple descriptions into detailed photography prompts + - Full GPU mode (≥7.5GB free): ~26-30 tokens/second + - CPU Offload mode (<7.5GB free): ~1-2 tokens/second, coexists with other models +- **Template System**: Select from pre-made prompt templates in the `Prompt` folder or use custom prompts - **Bilingual Support**: Handles both Chinese and English inputs - **Automatic Config Download**: Downloads required model configuration files from HuggingFace -- **Think Tag Removal**: Automatically removes model reasoning process from output +- **Think Tag Removal**: Automatically removes `...` reasoning tags from output ### Requirements - **GPU**: NVIDIA GPU with CUDA support (12GB+ recommended, works with less using CPU offload) -- **Model File**: qwen_3_4b.safetensors in `ComfyUI/models/text_encoders/` +- **Model Files**: + - **Required**: `qwen_3_4b.safetensors` (or similar Qwen3-4B safetensors file) + - **Location**: Place in either `ComfyUI/models/text_encoders/` or `ComfyUI/models/clip/` - **Python Packages**: transformers, safetensors, torch ### Model Setup -1. Download `qwen_3_4b.safetensors` from [Hugging Face](https://huggingface.co/Qwen/Qwen3-4B) -2. Place the file in `ComfyUI/models/text_encoders/` -3. First run will automatically download configuration files +1. Download Qwen3-4B safetensors from [Hugging Face](https://huggingface.co/Qwen/Qwen3-4B) + - Recommended filename: `qwen_3_4b.safetensors` +2. Place the file in `ComfyUI/models/text_encoders/` or `ComfyUI/models/clip/` +3. First run will automatically download configuration files to a `_config` subfolder ### Parameters | Parameter | Type | Default | Description | |-----------|------|---------|-------------| -| `user_prompt` | STRING | "一個女孩在咖啡廳" | Simple scene description | -| `system_prompt` | STRING | (see below) | Professional photography optimization prompt | -| `max_new_tokens` | INT | 2048 | Maximum length of generated prompt | +| `user_prompt` | STRING | "A girl in a coffee shop" | Your input text/description | +| `prompt_template` | COMBO | "Custom" | Select a template from the Prompt folder or use "Custom" | +| `system_prompt` | STRING | "" | Custom system prompt (used when template is "Custom") | +| `max_new_tokens` | INT | 2048 | Maximum length of generated text | | `temperature` | FLOAT | 0.7 | Creativity level (0.0-2.0) | | `do_sample` | BOOLEAN | True | Enable sampling for varied outputs | | `top_p` | FLOAT | 0.9 | Nucleus sampling threshold | | `top_k` | INT | 50 | Top-k sampling parameter | -### Default System Prompt +### Template System -The node uses a specialized system prompt optimized for generating professional photography descriptions: - -``` -You are a professional photography prompt optimization expert. Transform simple scene descriptions into detailed, professional photography prompts. - -Include these elements: -1. Subject Description: Detailed main subject (person, object, scene) -2. Environment Details: Surrounding environment, background, atmosphere -3. Lighting Effects: Light type, direction, contrast, color temperature -4. Camera Settings: Perspective, depth of field, focal length -5. Composition: Layout, foreground/midground/background -6. Color Atmosphere: Main colors, color matching, saturation -7. Texture Details: Materials, textures, detail expression -8. Mood Atmosphere: Overall atmosphere, emotional expression - -Output Format: -- Use English for professional photography terms -- Separate elements with commas -- Ensure descriptions are specific and visual -- Length: 150-300 English words -``` +The node supports customizable prompt templates stored in the `Prompt` folder as `.md` files: +- **Custom**: Use the `system_prompt` parameter directly +- **Template Files**: Select from available `.md` files in the `Prompt` folder +- Templates automatically replace the `system_prompt` parameter when selected ### Loading Strategies @@ -342,16 +330,19 @@ The node automatically selects the best loading strategy based on available GPU ### Usage Examples -#### Example 1: Simple Chinese Input +#### Example 1: Using a Template ``` -Input: "一個女孩在咖啡廳" +User Prompt: "一個女孩在咖啡廳" +Template: (photography template from Prompt folder) Output: "A young woman sitting by the window in a cozy coffee shop, warm afternoon sunlight streaming through large glass windows creating soft shadows, wearing casual outfit, holding a cup of coffee, wooden table with laptop and notebook, blurred background with other customers, shallow depth of field, bokeh effect, warm color temperature, golden hour lighting..." ``` -#### Example 2: English Input +#### Example 2: Custom System Prompt ``` -Input: "a cat sitting on a windowsill" -Output: "A calico cat sitting on a sunlit windowsill, its long fur catching golden-hour light from the right, creating soft shadows across its face and body, wearing a quiet expression of peaceful solitude, surrounded by indoor plants and a wooden bookshelf in the background, shallow depth of field with bokeh-like blur..." +User Prompt: "a cat sitting on a windowsill" +Template: "Custom" +System Prompt: "Describe the scene in poetic detail" +Output: "A calico cat sitting on a sunlit windowsill, its long fur catching golden-hour light from the right, creating soft shadows across its face and body, wearing a quiet expression of peaceful solitude, surrounded by indoor plants and a wooden bookshelf in the background..." ``` ### Memory Management @@ -375,8 +366,9 @@ The node includes intelligent memory management: ### Troubleshooting **Model not found** -- Ensure `qwen_3_4b.safetensors` is in `models/text_encoders/` -- Check filename matches exactly (case-sensitive) +- Ensure a Qwen safetensors file is in `models/text_encoders/` or `models/clip/` +- Recommended filename: `qwen_3_4b.safetensors` (containing "qwen", "3", and "4b") +- The node automatically searches for compatible Qwen model files **Out of memory** - Node will automatically use CPU offload @@ -643,42 +635,54 @@ NumberListGenerator 節點可根據自訂參數創建數字列表,支援有序 ### 概述 -**QwenGPUInference** 節點是一個 AI 驅動的照片提示詞優化器,將簡單的場景描述轉換為詳細、專業的攝影提示詞。使用 Qwen3-4B 語言模型,具備智能 GPU 記憶體管理和 CPU Offload 支援,可與 ComfyUI 無縫整合。 +**QwenGPUInference** 節點是一個 AI 驅動的文本生成節點,使用 Qwen3-4B 語言模型,具備智能 GPU 記憶體管理和 CPU Offload 支援,可與 ComfyUI 無縫整合。可使用自訂模板將簡單描述轉換為詳細、專業的提示詞。 ### 功能特色 -- **自動模型偵測**:自動在 text_encoders 資料夾中尋找 qwen_3_4b.safetensors +- **自動模型偵測**:自動在 `text_encoders` 或 `clip` 資料夾中尋找 Qwen safetensors 檔案 - **智能記憶體管理**:自動檢測 GPU 記憶體並選擇最佳載入策略 -- **CPU Offload 支援**:可與其他 ComfyUI 模型共存,無記憶體衝突 -- **專業提示詞生成**:將簡單描述轉換為詳細的攝影提示詞 + - 完全 GPU 模式(可用 ≥7.5GB):~26-30 tokens/秒 + - CPU Offload 模式(可用 <7.5GB):~1-2 tokens/秒,可與其他模型共存 +- **模板系統**:從 `Prompt` 資料夾選擇預製模板或使用自訂提示詞 - **雙語支援**:處理中文和英文輸入 - **自動配置下載**:從 HuggingFace 自動下載所需的模型配置檔案 -- **思考標籤移除**:自動移除模型推理過程 +- **思考標籤移除**:自動移除 `...` 推理標籤 ### 需求 - **GPU**:支援 CUDA 的 NVIDIA GPU(建議 12GB+,記憶體較少時使用 CPU offload) -- **模型檔案**:qwen_3_4b.safetensors 放在 `ComfyUI/models/text_encoders/` +- **模型檔案**: + - **必需**:`qwen_3_4b.safetensors`(或類似的 Qwen3-4B safetensors 檔案) + - **位置**:放在 `ComfyUI/models/text_encoders/` 或 `ComfyUI/models/clip/` - **Python 套件**:transformers、safetensors、torch ### 模型設置 -1. 從 [Hugging Face](https://huggingface.co/Qwen/Qwen3-4B) 下載 `qwen_3_4b.safetensors` -2. 將檔案放在 `ComfyUI/models/text_encoders/` -3. 首次執行會自動下載配置檔案 +1. 從 [Hugging Face](https://huggingface.co/Qwen/Qwen3-4B) 下載 Qwen3-4B safetensors + - 建議檔名:`qwen_3_4b.safetensors` +2. 將檔案放在 `ComfyUI/models/text_encoders/` 或 `ComfyUI/models/clip/` +3. 首次執行會自動下載配置檔案至 `_config` 子資料夾 ### 參數說明 | 參數 | 類型 | 預設值 | 說明 | |------|------|--------|------| -| `user_prompt` | STRING | "一個女孩在咖啡廳" | 簡單場景描述 | -| `system_prompt` | STRING | (見下方) | 專業攝影優化提示詞 | -| `max_new_tokens` | INT | 2048 | 生成提示詞的最大長度 | +| `user_prompt` | STRING | "A girl in a coffee shop" | 您的輸入文字/描述 | +| `prompt_template` | COMBO | "Custom" | 從 Prompt 資料夾選擇模板或使用 "Custom" | +| `system_prompt` | STRING | "" | 自訂系統提示詞(模板為 "Custom" 時使用)| +| `max_new_tokens` | INT | 2048 | 生成文字的最大長度 | | `temperature` | FLOAT | 0.7 | 創意程度 (0.0-2.0) | | `do_sample` | BOOLEAN | True | 啟用採樣以產生變化輸出 | | `top_p` | FLOAT | 0.9 | Nucleus 採樣閾值 | | `top_k` | INT | 50 | Top-k 採樣參數 | +### 模板系統 + +節點支援儲存在 `Prompt` 資料夾中的自訂提示詞模板(`.md` 檔案): +- **Custom**:直接使用 `system_prompt` 參數 +- **模板檔案**:從 `Prompt` 資料夾中的 `.md` 檔案選擇 +- 選擇模板時會自動取代 `system_prompt` 參數 + ### 載入策略 節點會根據可用 GPU 記憶體自動選擇最佳載入策略: @@ -706,16 +710,19 @@ NumberListGenerator 節點可根據自訂參數創建數字列表,支援有序 ### 使用範例 -#### 範例 1:簡單中文輸入 +#### 範例 1:使用模板 ``` -輸入: "一個女孩在咖啡廳" +使用者提示詞: "一個女孩在咖啡廳" +模板: (Prompt 資料夾中的攝影模板) 輸出: "A young woman sitting by the window in a cozy coffee shop, warm afternoon sunlight streaming through large glass windows creating soft shadows, wearing casual outfit, holding a cup of coffee, wooden table with laptop and notebook, blurred background with other customers, shallow depth of field, bokeh effect, warm color temperature, golden hour lighting..." ``` -#### 範例 2:英文輸入 +#### 範例 2:自訂系統提示詞 ``` -輸入: "a cat sitting on a windowsill" -輸出: "A calico cat sitting on a sunlit windowsill, its long fur catching golden-hour light from the right, creating soft shadows across its face and body, wearing a quiet expression of peaceful solitude, surrounded by indoor plants and a wooden bookshelf in the background, shallow depth of field with bokeh-like blur..." +使用者提示詞: "a cat sitting on a windowsill" +模板: "Custom" +系統提示詞: "用詩意的細節描述場景" +輸出: "A calico cat sitting on a sunlit windowsill, its long fur catching golden-hour light from the right, creating soft shadows across its face and body, wearing a quiet expression of peaceful solitude, surrounded by indoor plants and a wooden bookshelf in the background..." ``` ### 記憶體管理 @@ -739,8 +746,9 @@ NumberListGenerator 節點可根據自訂參數創建數字列表,支援有序 ### 疑難排解 **找不到模型** -- 確保 `qwen_3_4b.safetensors` 在 `models/text_encoders/` -- 檢查檔案名稱完全符合(區分大小寫) +- 確保 Qwen safetensors 檔案在 `models/text_encoders/` 或 `models/clip/` +- 建議檔名:`qwen_3_4b.safetensors`(包含 "qwen"、"3" 和 "4b") +- 節點會自動搜尋相容的 Qwen 模型檔案 **記憶體不足** - 節點會自動使用 CPU offload