From e91a1ce2c6cc1947757a6aac5a00e24da17cd4e3 Mon Sep 17 00:00:00 2001 From: Zuellni <123005779+Zuellni@users.noreply.github.com> Date: Fri, 14 Jun 2024 10:57:28 +0200 Subject: [PATCH 1/7] Switch to dynamic generator --- exllama.py | 38 +++++++++++++++++++++----------------- requirements.txt | 3 ++- 2 files changed, 23 insertions(+), 18 deletions(-) diff --git a/exllama.py b/exllama.py index 7661f64..e988f0b 100644 --- a/exllama.py +++ b/exllama.py @@ -31,7 +31,7 @@ class Loader: return { "required": { "model": (models, {"default": default}), - "cache_bits": ((4, 8, 16), {"default": 16}), + "cache_bits": ((4, 6, 8, 16), {"default": 16}), "max_seq_len": ("INT", {"default": 2048, "max": 2**20}), }, } @@ -75,14 +75,16 @@ class Loader: self.cache = ( ExLlamaV2Cache_Q4(self.model, lazy=True) if self.cache_bits == 4 - else ExLlamaV2Cache_8bit(self.model, lazy=True) + else ExLlamaV2Cache_Q6(self.model, lazy=True) + if self.cache_bits == 6 + else ExLlamaV2Cache_Q8(self.model, lazy=True) if self.cache_bits == 8 else ExLlamaV2Cache(self.model, lazy=True) ) self.tokenizer = ExLlamaV2Tokenizer(self.config) self.model.load_autosplit(self.cache, callback=lambda _, __: progress.update(1)) - self.generator = ExLlamaV2StreamingGenerator(self.model, self.cache, self.tokenizer) + self.generator = ExLlamaV2DynamicGenerator(self.model, self.cache, self.tokenizer) def unload(self): if hasattr(self, "model") and self.model: @@ -155,6 +157,7 @@ class Generator: model.unload() model.load() + random.seed(seed) input = model.tokenizer.encode(text, encode_special_tokens=True) input_len = input.shape[-1] max_len = model.config.max_seq_len - input_len @@ -164,10 +167,7 @@ class Generator: max_tokens = max_len if single_line: - stop.append(model.tokenizer.newline_token_id) - - model.generator.set_stop_conditions(stop) - random.seed(seed) + stop.append("\n") settings = ExLlamaV2Sampler.Settings() settings.temperature = temperature @@ -179,21 +179,25 @@ class Generator: settings.token_repetition_penalty = repetition_penalty settings.temperature_last = temperature_last - start = time() - model.generator.begin_stream_ex(input, settings) + job = ExLlamaV2DynamicJob(input, max_new_tokens=max_tokens, stop_conditions=stop) + model.generator.enqueue(job) + progress = ProgressBar(max_tokens) + start = time() + eos = False - output = "" + chunks = [] tokens = 0 - while not eos and tokens < max_tokens: - response = model.generator.stream_ex() - output += response["chunk"] - eos = response["eos"] - progress.update(1) - tokens += 1 + while not eos: + for response in model.generator.iterate(): + if response["stage"] == "streaming": + chunks.append(response.get("text", "")) + eos = response["eos"] + progress.update(1) + tokens += 1 - output = output.strip() + output = "".join(chunks).strip() total = round(time() - start, 2) speed = round(tokens / total, 2) diff --git a/requirements.txt b/requirements.txt index edbf7cb..66cebb5 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1 +1,2 @@ -exllamav2>=0.0.17; platform_system == "Linux" +exllamav2; platform_system == "Linux" +flash-attn; platform_system == "Linux" From f98870361475ed2c43de096c056fcec66fd40dfc Mon Sep 17 00:00:00 2001 From: Zuellni <123005779+Zuellni@users.noreply.github.com> Date: Fri, 14 Jun 2024 11:33:10 +0200 Subject: [PATCH 2/7] Add custom stop conditions --- exllama.py | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/exllama.py b/exllama.py index e988f0b..e3ce6e8 100644 --- a/exllama.py +++ b/exllama.py @@ -1,4 +1,5 @@ import gc +import json import random from pathlib import Path from time import time @@ -106,7 +107,7 @@ class Generator: "required": { "model": ("EXL_MODEL",), "unload": ("BOOLEAN", {"default": False}), - "single_line": ("BOOLEAN", {"default": False}), + "stop_conditions": ("STRING", {"default": r'["\n"]'}), "max_tokens": ("INT", {"default": 128, "max": 2**20}), "temperature": ("FLOAT", {"default": 1, "max": 5, "step": 0.01}), "top_k": ("INT", {"max": 200}), @@ -134,7 +135,7 @@ class Generator: self, model, unload, - single_line, + stop_conditions, max_tokens, temperature, top_k, @@ -149,7 +150,7 @@ class Generator: info=None, id=None, ): - if not text: + if not text.strip(): return ("",) if unload: @@ -166,8 +167,9 @@ class Generator: if not max_tokens or max_tokens > max_len: max_tokens = max_len - if single_line: - stop.append("\n") + if stop_conditions.strip(): + stop_conditions = json.loads(stop_conditions) + stop.extend(stop_conditions) settings = ExLlamaV2Sampler.Settings() settings.temperature = temperature @@ -192,8 +194,9 @@ class Generator: while not eos: for response in model.generator.iterate(): if response["stage"] == "streaming": - chunks.append(response.get("text", "")) + chunk = response.get("text", "") eos = response["eos"] + chunks.append(chunk) progress.update(1) tokens += 1 From 4754f78c70911eebde5b846db076f39395f35681 Mon Sep 17 00:00:00 2001 From: Zuellni <123005779+Zuellni@users.noreply.github.com> Date: Fri, 14 Jun 2024 11:50:22 +0200 Subject: [PATCH 3/7] Add option to enable/disable flash attention and fast tensors --- exllama.py | 23 ++++++++++++++++++----- 1 file changed, 18 insertions(+), 5 deletions(-) diff --git a/exllama.py b/exllama.py index e3ce6e8..37ef347 100644 --- a/exllama.py +++ b/exllama.py @@ -33,6 +33,8 @@ class Loader: "required": { "model": (models, {"default": default}), "cache_bits": ((4, 6, 8, 16), {"default": 16}), + "fast_tensors": ("BOOLEAN", {"default": True}), + "flash_attention": ("BOOLEAN", {"default": True}), "max_seq_len": ("INT", {"default": 2048, "max": 2**20}), }, } @@ -43,10 +45,12 @@ class Loader: RETURN_NAMES = ("MODEL",) RETURN_TYPES = ("EXL_MODEL",) - def setup(self, model, cache_bits, max_seq_len): + def setup(self, model, cache_bits, fast_tensors, flash_attention, max_seq_len): self.unload() self.cache_bits = cache_bits self.config = ExLlamaV2Config(__class__._MODELS[model]) + self.config.fasttensors = fast_tensors + self.config.no_flash_attn = not flash_attention if max_seq_len: self.config.max_seq_len = max_seq_len @@ -85,7 +89,13 @@ class Loader: self.tokenizer = ExLlamaV2Tokenizer(self.config) self.model.load_autosplit(self.cache, callback=lambda _, __: progress.update(1)) - self.generator = ExLlamaV2DynamicGenerator(self.model, self.cache, self.tokenizer) + + self.generator = ExLlamaV2DynamicGenerator( + model=self.model, + cache=self.cache, + tokenizer=self.tokenizer, + paged=not self.config.no_flash_attn, + ) def unload(self): if hasattr(self, "model") and self.model: @@ -181,12 +191,15 @@ class Generator: settings.token_repetition_penalty = repetition_penalty settings.temperature_last = temperature_last - job = ExLlamaV2DynamicJob(input, max_new_tokens=max_tokens, stop_conditions=stop) - model.generator.enqueue(job) + job = ExLlamaV2DynamicJob( + input_ids=input, + max_new_tokens=max_tokens, + stop_conditions=stop, + ) progress = ProgressBar(max_tokens) + model.generator.enqueue(job) start = time() - eos = False chunks = [] tokens = 0 From f54093078ee61563922a34b2a718aa45ff54a4bd Mon Sep 17 00:00:00 2001 From: Zuellni <123005779+Zuellni@users.noreply.github.com> Date: Fri, 14 Jun 2024 13:21:42 +0200 Subject: [PATCH 4/7] Clean up README --- README.md | 19 +++---------------- 1 file changed, 3 insertions(+), 16 deletions(-) diff --git a/README.md b/README.md index 6d94885..e6d71b7 100644 --- a/README.md +++ b/README.md @@ -7,29 +7,16 @@ Clone the repository to `custom_nodes`: git clone https://github.com/Zuellni/ComfyUI-ExLlama-Nodes custom_nodes/ComfyUI-ExLlamaV2-Nodes ``` -Install the requirements: +Install requirements, use wheels for [ExLlamaV2](https://github.com/turboderp/exllamav2/releases/latest) and [Flash Attention](https://github.com/bdashore3/flash-attention/releases/latest) on Windows: ``` pip install -r custom_nodes/ComfyUI-ExLlamaV2-Nodes/requirements.txt ``` -On Windows, install one of the precompiled [wheels](https://github.com/turboderp/exllamav2/releases/latest) instead: -``` -pip install https://github.com/turboderp/exllamav2/releases/download/v0.0.xx/exllamav2-0.0.xx+cuXXX-cpXXX-cpXXX-win_amd64.whl -``` - -Check which one you need with: -``` -python -c "import sys, torch; print(f'cu{torch.version.cuda.replace('.', '')}-cp{sys.version_info[0]}{sys.version_info[1]}')" -``` - -> [!CAUTION] -> If you see errors related to ExLlamaV2 while loading the nodes, try to install it following the [official instructions](https://github.com/turboderp/exllamav2#installation). - ## Usage -Only EXL2, 4-bit GPTQ, and unquantized HF models are supported. You can find them on [Hugging Face](https://huggingface.co). See the model card in each repository for details on instruction formats. +Only EXL2, 4-bit GPTQ, and unquantized HF models are supported. You can find them on [Hugging Face](https://huggingface.co). To use a model with the nodes, you should clone its repository with git or manually download all the files and place them in `models/llm`. -For example, if you'd like to download the 6-bit [Llama-3-8B-Instruct](https://huggingface.co/turboderp/Llama-3-8B-Instruct-exl2), use the following command: +For example, if you want to download the 6-bit [Llama-3-8B-Instruct](https://huggingface.co/turboderp/Llama-3-8B-Instruct-exl2), use the following command: ``` git install lfs git clone https://huggingface.co/turboderp/Llama-3-8B-Instruct-exl2 -b 6.0bpw models/llm/Llama-3-8B-Instruct-exl2-6.0bpw From cac9d1fdf42126246140e6134510edd107c7b6ef Mon Sep 17 00:00:00 2001 From: Zuellni <123005779+Zuellni@users.noreply.github.com> Date: Fri, 14 Jun 2024 13:24:38 +0200 Subject: [PATCH 5/7] Update README.md --- README.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index e6d71b7..efa8079 100644 --- a/README.md +++ b/README.md @@ -13,14 +13,15 @@ pip install -r custom_nodes/ComfyUI-ExLlamaV2-Nodes/requirements.txt ``` ## Usage -Only EXL2, 4-bit GPTQ, and unquantized HF models are supported. You can find them on [Hugging Face](https://huggingface.co). +Only EXL2, 4-bit GPTQ and FP16 HF models are supported. You can find them on [Hugging Face](https://huggingface.co). -To use a model with the nodes, you should clone its repository with git or manually download all the files and place them in `models/llm`. +To use a model with the nodes, you should clone its repository with `git` or manually download all the files and place them in `models/llm`. For example, if you want to download the 6-bit [Llama-3-8B-Instruct](https://huggingface.co/turboderp/Llama-3-8B-Instruct-exl2), use the following command: ``` git install lfs git clone https://huggingface.co/turboderp/Llama-3-8B-Instruct-exl2 -b 6.0bpw models/llm/Llama-3-8B-Instruct-exl2-6.0bpw ``` + > [!TIP] > You can add your own `llm` path to the [extra_model_paths.yaml](https://github.com/comfyanonymous/ComfyUI/blob/master/extra_model_paths.yaml.example) file and put the models there instead. From 51728c01025ff82a90109d1a3edba4ae244bac59 Mon Sep 17 00:00:00 2001 From: Zuellni <123005779+Zuellni@users.noreply.github.com> Date: Fri, 14 Jun 2024 13:28:48 +0200 Subject: [PATCH 6/7] Update requirements.txt --- requirements.txt | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/requirements.txt b/requirements.txt index 66cebb5..04c8177 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,2 +1,2 @@ -exllamav2; platform_system == "Linux" -flash-attn; platform_system == "Linux" +exllamav2>=0.1.5; platform_system == "Linux" +flash-attn>=2.5.7; platform_system == "Linux" From 5a90464b6ceb445e06e380e82623932046346091 Mon Sep 17 00:00:00 2001 From: Zuellni <123005779+Zuellni@users.noreply.github.com> Date: Fri, 14 Jun 2024 18:48:05 +0200 Subject: [PATCH 7/7] Update README.md --- README.md | 40 ++++++++++++++++++++++++++-------------- 1 file changed, 26 insertions(+), 14 deletions(-) diff --git a/README.md b/README.md index efa8079..a4577ec 100644 --- a/README.md +++ b/README.md @@ -2,18 +2,20 @@ A simple local text generator for [ComfyUI](https://github.com/comfyanonymous/ComfyUI) using [ExLlamaV2](https://github.com/turboderp/exllamav2). ## Installation -Clone the repository to `custom_nodes`: +Clone the repository to `custom_nodes` and install the requirements: ``` git clone https://github.com/Zuellni/ComfyUI-ExLlama-Nodes custom_nodes/ComfyUI-ExLlamaV2-Nodes -``` - -Install requirements, use wheels for [ExLlamaV2](https://github.com/turboderp/exllamav2/releases/latest) and [Flash Attention](https://github.com/bdashore3/flash-attention/releases/latest) on Windows: -``` pip install -r custom_nodes/ComfyUI-ExLlamaV2-Nodes/requirements.txt ``` +Use wheels for [ExLlamaV2](https://github.com/turboderp/exllamav2/releases/latest) and [Flash Attention](https://github.com/bdashore3/flash-attention/releases/latest) on Windows: +``` +pip install exllamav2-X.X.X+cuXXX.torch2.X.X-cp3XX-cp3XX-win_amd64.whl +pip install flash_attn-X.X.X+cuXXX.torch2.X.X-cp3XX-cp3XX-win_amd64.whl +``` + ## Usage -Only EXL2, 4-bit GPTQ and FP16 HF models are supported. You can find them on [Hugging Face](https://huggingface.co). +Only EXL2, 4-bit GPTQ and unquantized models are supported. You can find them on [Hugging Face](https://huggingface.co). To use a model with the nodes, you should clone its repository with `git` or manually download all the files and place them in `models/llm`. For example, if you want to download the 6-bit [Llama-3-8B-Instruct](https://huggingface.co/turboderp/Llama-3-8B-Instruct-exl2), use the following command: @@ -34,26 +36,36 @@ git clone https://huggingface.co/turboderp/Llama-3-8B-Instruct-exl2 -b 6.0bpw mo
8.0.0 will default to config.0 will default to model config.["\n"] to stop on newline. Leave empty to only stop on eos token.[a], with their values.[a], with their values.