diff --git a/requirements.txt b/requirements.txt index f14b86e..24856cb 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,19 +1,19 @@ # ComfyUI-QwenVL requirements -# Note: Qwen3-VL models require transformers >= 4.57.0 +# Note: Qwen3-VL and Qwen2.5-VL models require transformers >= 4.57.0 (for Vision2Seq support) +# Torch >= 2.1.0 is recommended for torch.compile and SDPA/Flash-Attn integration -transformers -torch -huggingface-hub +torch>=2.1.0 +transformers>=4.46.0 +huggingface-hub>=0.23.0 +bitsandbytes>=0.43.0 +accelerate>=0.33.0 psutil numpy Pillow opencv-python -bitsandbytes -accelerate -# For running GGUF models with vision support +# Optional (for FlashAttention v2 optimization) +flash-attn>=2.5.6; platform_system=="Linux" and extra=="cuda" -# The [server] extra installs necessary components for multimodal models +# For running GGUF models with vision support (planned feature) # llama-cpp-python[server] - -