diff --git a/data/custom_prompts/zimage_vision_analysis.txt b/data/custom_prompts/zimage_vision_analysis.txt index 1ff0bed..d5e70a3 100644 --- a/data/custom_prompts/zimage_vision_analysis.txt +++ b/data/custom_prompts/zimage_vision_analysis.txt @@ -1,13 +1,37 @@ -Analyze this image in detail and output a JSON object describing what you see. Include any relevant details about: +Analyze the image strictly as a Director of Photography and Colorist. Output a JSON object. -- Subject(s): People, characters, objects, animals - describe their appearance, clothing, expressions, poses -- Setting: Location, environment, background elements -- Style: Art style, photography style, lighting, color palette, mood -- Composition: Camera angle, framing, focal points -- Any text visible in the image -- Notable details or unique elements - -Output ONLY valid JSON. Structure the JSON however makes sense for the image content. Be thorough and descriptive. - -CRITICAL: Output ONLY the JSON object. No explanations, no markdown code blocks, no extra text. Start with { and end with }. +CRITICAL: You must explicitly describe the LIGHT INTENSITY, EXPOSURE, and CONTRAST in the "lighting_dynamics" section. +{ + "subject_content": { + "main_subject": "Detailed description of character/object including pose and expression", + "clothing_and_details": "Textures, fabrics, colors of attire", + "action": "What is happening" + }, + "setting_and_composition": { + "location": "Environment details, etc", + "framing_style": "e.g., Dutch Angle, Symmetrical, Wide Angle distortion, Telephoto compression, etc" + "depth_of_field": "Deep focus or Shallow (Bokeh), etc", + "subject_position": "Precise location in frame (e.g., Dead Center, Rule of Thirds Left, Bottom Right corner, Extreme Foreground, To the left of the frame, to the right of the frame, in the back left, in the back right), etc", + "subject_scale": "How large the subject is (e.g., Tiny silhouette, Full Body, Extreme Close-up filling frame, Half Body, Upper body), etc", + "surrounding_elements": "Description of what occupies the specific areas where the subject IS NOT (e.g., 'To the right is an empty concrete wall', 'The background is a blurred cityscape')", + } + "technical_aesthetic": { + "lighting_dynamics": { + "intensity": "e.g., Blindingly bright, Dim, Subdued, Harsh, Soft glow, etc", + "overall_key": "High-key (bright/optimistic) or Low-key (dark/shadowy/moody), etc", + "contrast_level": "High contrast (deep blacks/bright whites) or Low contrast (flat/grey/washed out), etc", + "exposure": "Balanced, Overexposed (blown highlights), or Underexposed (crushed shadows), etc" + "colors": "What colors are on the lights in the scene, etc" + }, + "lighting_setup": { + "style": "Natural, Studio, Volumetric/Godrays, Chiaroscuro, Neon, etc", + "direction": "Backlit (silhouette), Side-lit, Top-down, Rim-lighting, etc" + }, + "color_grade": { + "palette": "Dominant hex codes or color names", + "grading_style": "Teal & Orange, Bleach Bypass, Technicolor, Muted, Vibrant, Monochromatic, etc", + "film_character": "Grain, Halation, Vintage, Clean Digital,e tc" + } + } +} \ No newline at end of file diff --git a/data/custom_prompts/zimage_vision_system.txt b/data/custom_prompts/zimage_vision_system.txt index b3e6b67..af39bd8 100644 --- a/data/custom_prompts/zimage_vision_system.txt +++ b/data/custom_prompts/zimage_vision_system.txt @@ -1,2 +1,10 @@ -Generate an image based on the detailed visual specification provided. Follow the description precisely, paying attention to all visual details, composition, lighting, and style mentioned. +You are an expert Image Generative AI specialized in replicating visual atmosphere. +Your goal is to recreate an image based on the provided visual specification. + +CRITICAL INSTRUCTION: You must strictly adhere to the 'lighting_dynamics' and 'technical_aesthetic'. +1. If the analysis says "Low-key" or "Underexposed", the generated image MUST be dark and shadowy. +2. If the analysis says "High-key" or "Overexposed", the generated image MUST be bright and light. +3. Match the specific contrast levels (High vs Low) and color grading (LUT) described. + +Prioritize the lighting mood and color science over minor background details. \ No newline at end of file diff --git a/nodes/qwenvl/zimage_vision.py b/nodes/qwenvl/zimage_vision.py index e55d6cb..28ccbb0 100644 --- a/nodes/qwenvl/zimage_vision.py +++ b/nodes/qwenvl/zimage_vision.py @@ -60,7 +60,7 @@ class QwenVLZImageVision: "system_prompt_file": (prompt_files, {"default": default_system}), "user_mod_file": (prompt_files, {"default": default_user_mod}), "max_tokens": ("INT", {"default": 4096, "min": 512, "max": 8192}), - "temperature": ("FLOAT", {"default": 0.7, "min": 0.1, "max": 1.0}), + "temperature": ("FLOAT", {"default": 0.5, "min": 0.1, "max": 1.0}), "keep_model_loaded": ("BOOLEAN", {"default": True}), "include_system_prompt": ("BOOLEAN", {"default": True}), "include_think_block": ("BOOLEAN", {"default": False}), @@ -274,7 +274,7 @@ CRITICAL: Output ONLY the JSON object. No explanations, no markdown code blocks, def analyze_image(self, images, qwen_model="Qwen3-VL-4B-Instruct", prompt_file="(none)", system_prompt_file="(none)", user_mod_file="(none)", - max_tokens=4096, temperature=0.7, keep_model_loaded=True, + max_tokens=4096, temperature=0.5, keep_model_loaded=True, include_system_prompt=True, include_think_block=True, strip_quotes=False, custom_analysis_prompt="", user_modification="", custom_system_prompt=""): try: @@ -342,6 +342,7 @@ CRITICAL: Output ONLY the JSON object. No explanations, no markdown code blocks, content = tokenizer.decode(outputs[0, input_len:], skip_special_tokens=True).strip() print(f"✅ Z-Image Vision: Received response ({len(content)} chars)") + print(f"📝 Raw response from Qwen model:\n{content}\n{'='*60}") if not content: print("⚠️ Model returned empty response")