diff --git a/README.md b/README.md index 6b16545..0781f55 100644 --- a/README.md +++ b/README.md @@ -2,17 +2,27 @@ A simple text generator for [ComfyUI](https://github.com/comfyanonymous/ComfyUI) utilizing [ExLlamaV2](https://github.com/turboderp/exllamav2). ## Installation -Navigate to the root ComfyUI directory, clone the repository to `custom_nodes` and install dependencies: +Navigate to the root ComfyUI directory and clone the repository to `custom_nodes`: ``` git clone https://github.com/Zuellni/ComfyUI-ExLlama-Nodes custom_nodes/ComfyUI-ExLlama-Nodes -pip install -r custom_nodes/ComfyUI-ExLlama-Nodes/requirements.txt ``` -Optionally, you can install [flash-attention](https://github.com/Dao-AILab/flash-attention) by uncommenting relevant lines in the requirements file. It should lower VRAM usage but your mileage may vary. + +Install the requirements depending on your system: +``` +pip install -r custom_nodes/ComfyUI-ExLlama-Nodes/requirements-VERSION.txt +``` + +`requirements-no-wheels.txt` contains JIT version of ExLlamaV2 and [Flash Attention](https://github.com/Dao-AILab/flash-attention).
+`requirements-torch-21.txt` contains Windows wheels for Python 3.11, Torch 2.1, CUDA 12.1.
+`requirements-torch-22.txt` contains Windows wheels for Python 3.11, Torch 2.2, CUDA 12.1. + +Check what you need with: +``` +python -c "import platform; import torch; print(f'Python {platform.python_version()}, Torch {torch.__version__}, CUDA {torch.version.cuda}')" +``` + > [!CAUTION] -> Manual installation without any managers is recommended since the requirements depend on your system.
-> Wheels included in the requirements file match the `stable pytorch 2.1 cu121` portable build of ComfyUI.
-> If you see any ExLlama-related errors while the nodes are loading, try to install it following the [official instructions](https://github.com/turboderp/exllamav2#installation).
-> Keep in mind that wheels >= `0.0.13` require `pytorch 2.2`. +> If none of the included wheels work for you or there are any ExLlama-related errors while the nodes are loading, try to install it manually following the [official instructions](https://github.com/turboderp/exllamav2#installation). Keep in mind that wheels >= `0.0.13` require Torch 2.2. ## Usage Only EXL2 and 4-bit GPTQ models are supported. You can find a lot of them on [Hugging](https://huggingface.co/LoneStriker) [Face](https://huggingface.co/TheBloke). Refer to the model card in each repository for details about quant differences and instruction formats. diff --git a/exllama.py b/exllama.py index a3be042..3afe677 100644 --- a/exllama.py +++ b/exllama.py @@ -106,13 +106,13 @@ class Generator: "single_line": ("BOOLEAN", {"default": False}), "temperature_last": ("BOOLEAN", {"default": True}), "max_tokens": ("INT", {"default": 128, "max": 2**16}), - "temperature": ("FLOAT", {"default": 1, "max": 2, "step": 0.01}), + "temperature": ("FLOAT", {"default": 1, "max": 5, "step": 0.01}), "top_k": ("INT", {"max": 200}), "top_a": ("FLOAT", {"max": 1, "step": 0.01}), "min_p": ("FLOAT", {"max": 1, "step": 0.01}), - "top_p": ("FLOAT", {"max": 1, "step": 0.01}), + "top_p": ("FLOAT", {"default": 1, "max": 1, "step": 0.01}), "typical": ("FLOAT", {"default": 1, "max": 1, "step": 0.01}), - "penalty": ("FLOAT", {"default": 1, "min": 1, "max": 2, "step": 0.01}), + "penalty": ("FLOAT", {"default": 1, "min": 1, "max": 3, "step": 0.01}), "seed": ("INT", {"max": 2**64 - 1}), "text": ("STRING", {"multiline": True}), }, @@ -175,7 +175,7 @@ class Generator: settings.token_repetition_penalty = penalty start = time() - model.generator.begin_stream(input, settings, token_healing=True) + model.generator.begin_stream(input, settings) progress = ProgressBar(max_tokens) eos = False output = "" diff --git a/requirements-no-wheels.txt b/requirements-no-wheels.txt new file mode 100644 index 0000000..744601d --- /dev/null +++ b/requirements-no-wheels.txt @@ -0,0 +1,2 @@ +exllamav2 +flash-attn diff --git a/requirements-torch-21.txt b/requirements-torch-21.txt new file mode 100644 index 0000000..44430f0 --- /dev/null +++ b/requirements-torch-21.txt @@ -0,0 +1,2 @@ +https://github.com/turboderp/exllamav2/releases/download/v0.0.12/exllamav2-0.0.12+cu121-cp311-cp311-win_amd64.whl +https://github.com/bdashore3/flash-attention/releases/download/v2.4.2/flash_attn-2.4.2+cu122torch2.1.2cxx11abiFALSE-cp311-cp311-win_amd64.whl diff --git a/requirements-torch-22.txt b/requirements-torch-22.txt new file mode 100644 index 0000000..cd4bc5b --- /dev/null +++ b/requirements-torch-22.txt @@ -0,0 +1,2 @@ +https://github.com/turboderp/exllamav2/releases/download/0.0.13.post1/exllamav2-0.0.13.post1+cu121-cp311-cp311-win_amd64.whl +https://github.com/bdashore3/flash-attention/releases/download/v2.5.2/flash_attn-2.5.2+cu122torch2.2.0cxx11abiFALSE-cp311-cp311-win_amd64.whl diff --git a/requirements.txt b/requirements.txt deleted file mode 100644 index b8f4a1b..0000000 --- a/requirements.txt +++ /dev/null @@ -1,11 +0,0 @@ -# Linux -exllamav2; platform_system == "Linux" -# flash-attn; platform_system == "Linux" - -# Windows - Torch 2.1 -https://github.com/turboderp/exllamav2/releases/download/v0.0.12/exllamav2-0.0.12+cu121-cp311-cp311-win_amd64.whl; platform_system == "Windows" -# https://github.com/bdashore3/flash-attention/releases/download/v2.4.2/flash_attn-2.4.2+cu122torch2.1.2cxx11abiFALSE-cp311-cp311-win_amd64.whl; platform_system == "Windows" - -# Windows - Torch 2.2 -# https://github.com/turboderp/exllamav2/releases/download/0.0.13.post1/exllamav2-0.0.13.post1+cu121-cp311-cp311-win_amd64.whl; platform_system == "Windows" -# https://github.com/bdashore3/flash-attention/releases/download/v2.5.2/flash_attn-2.5.2+cu122torch2.2.0cxx11abiFALSE-cp311-cp311-win_amd64.whl; platform_system == "Windows"