From b15c313aacbfeeadac38dd17e9bafa5f2498db07 Mon Sep 17 00:00:00 2001 From: Zuellni <123005779+Zuellni@users.noreply.github.com> Date: Thu, 21 Sep 2023 11:39:50 +0200 Subject: [PATCH] Separate lora node and some fixes --- README.md | 17 +++++++------- __init__.py | 4 +++- nodes.py | 66 ++++++++++++++++++++++++++++++++++------------------- 3 files changed, 54 insertions(+), 33 deletions(-) diff --git a/README.md b/README.md index 2a3e83c..9af3d36 100644 --- a/README.md +++ b/README.md @@ -1,27 +1,28 @@ # ComfyUI ExLlama Nodes -A simple prompt generator for [ComfyUI](https://github.com/comfyanonymous/ComfyUI) utilizing [ExLlama](https://github.com/turboderp/exllama). +A simple prompt generator for [ComfyUI](https://github.com/comfyanonymous/ComfyUI) using [ExLlama](https://github.com/turboderp/exllama). ## Installation -Clone the repository to `custom_nodes` in your ComfyUI directory and install the dependencies: +Clone the repository to `custom_nodes` in your ComfyUI directory and install dependencies: ``` git clone https://github.com/Zuellni/ComfyUI-ExLlama-Nodes python -m pip install -r requirements.txt ``` -If you see any errors related to ExLlama while loading the nodes, you should manually install the wheel matching your system from [here](https://github.com/jllllll/exllama/releases/latest). -For example, on Windows with Python 3.11 and PyTorch CUDA 12.1, you would use: +If you see any ExLlama-related errors while loading, manually install the wheel matching your system from [here](https://github.com/jllllll/exllama/releases/latest). +For example, on Windows with Python 3.10 and PyTorch CUDA 11.7: ``` -python -m pip install https://github.com/jllllll/exllama/releases/download/0.0.17/exllama-0.0.17+cu121-cp311-cp311-win_amd64.whl +python -m pip install https://github.com/jllllll/exllama/releases/download/0.0.17/exllama-0.0.17+cu117-cp310-cp310-win_amd64.whl ``` ## Nodes Name | Description :--- | :--- -Loader | Loads 4-bit GPTQ Llama/2 models. You can find a lot of them on [Hugging Face](https://huggingface.co/TheBloke).
Clone the model repository or download all the files in it and place them in an empty directory, then specify the path in `model_dir`. The `model.safetensors` file won't work on its own.

ExLlama allocates memory based on `max_seq_len`. Lowering it is a good way to save on VRAM. It's currently not possible to [offload](https://github.com/turboderp/exllama/issues/177) the model to RAM. -Generator | Returns a `string` based on the given `prompt` for use with other nodes. Default values correspond to the `simple-1` preset from [text-generation-webui](https://github.com/oobabooga/text-generation-webui). ExLlama isn't [deterministic](https://github.com/turboderp/exllama/issues/201), so the outputs may differ slightly even with the same seed.

To load a LoRA specify the path to its directory in `lora_dir`. It should contain `adapter_model.bin` and `adapter_config.json`. +Loader | Used to load 4-bit GPTQ Llama/2 models. You can find a lot of them on [Hugging Face](https://huggingface.co/TheBloke).
Clone the model repository or download all the files in it and place them in an empty directory, then specify the path in `model_dir`. The `model.safetensors` file won't work on its own.

ExLlama allocates memory based on `max_seq_len`. Lowering it is a good way to save on VRAM. It's currently not possible to [offload](https://github.com/turboderp/exllama/issues/177) the model to RAM. +LoRA Loader | Used to load LoRAs. Specify the directory in `lora_dir`, it should contain `adapter_model.bin` and `adapter_config.json`. LoRA parameter count has to match the model. +Generator | Generates a `string` based on the given `prompt` for use with other nodes. Default values correspond to the `simple-1` preset from [text-generation-webui](https://github.com/oobabooga/text-generation-webui). ExLlama isn't [deterministic](https://github.com/turboderp/exllama/issues/201), so the outputs may differ even with the same seed. Previewer | Displays generated outputs in the UI. ## Workflow -The workflow below can be loaded directly in ComfyUI. Model used: [MythoLogic-Mini-7B](https://huggingface.co/TheBloke/MythoLogic-Mini-7B-GPTQ). +The workflow below can be opened in ComfyUI. Peak VRAM usage with SDXL around 10GB. Model: [MythoLogic-Mini-7B](https://huggingface.co/TheBloke/MythoLogic-Mini-7B-GPTQ). ![workflow](https://github.com/Zuellni/ComfyUI-ExLlama-Nodes/assets/123005779/c6821ba6-3a7a-4dd2-9852-372f79f63569) diff --git a/__init__.py b/__init__.py index eb7005d..8446058 100644 --- a/__init__.py +++ b/__init__.py @@ -1,13 +1,15 @@ -from .nodes import Generator, Loader, Previewer +from .nodes import Generator, Loader, Lora, Previewer NODE_CLASS_MAPPINGS = { "ZuellniExLlamaLoader": Loader, + "ZuellniExLlamaLoraLoader": Lora, "ZuellniExLlamaGenerator": Generator, "ZuellniExLlamaPreviewer": Previewer, } NODE_DISPLAY_NAME_MAPPINGS = { "ZuellniExLlamaLoader": "ExLlama Loader", + "ZuellniExLlamaLoraLoader": "ExLlama LoRA Loader", "ZuellniExLlamaGenerator": "ExLlama Generator", "ZuellniExLlamaPreviewer": "ExLlama Previewer", } diff --git a/nodes.py b/nodes.py index 9c7e3ee..c1d9bef 100644 --- a/nodes.py +++ b/nodes.py @@ -3,11 +3,13 @@ from platform import sys import torch from colorama import Fore -from comfy.model_management import soft_empty_cache from comfy.utils import ProgressBar -cu = "cu" + torch.version.cuda.replace(".", "") -cp = f"cp{sys.version_info.major}{sys.version_info.minor}" +if not torch.cuda.is_available(): + raise Exception(f"\n{Fore.RED}No CUDA detected. ExLlama doesn't support CPU.{Fore.RESET}") + +cuda = torch.version.cuda.replace(".", "") +pckg = f"cu{cuda}-cp{sys.version_info.major}{sys.version_info.minor}" try: from exllama.alt_generator import ExLlamaAltGenerator @@ -15,14 +17,14 @@ try: from exllama.model import ExLlama, ExLlamaCache, ExLlamaConfig from exllama.tokenizer import ExLlamaTokenizer except ModuleNotFoundError: - raise ModuleNotFoundError( - f"\n{Fore.RED}ExLlama not installed. Get {Fore.CYAN}{cu}-{cp}{Fore.RED} from\n" - f"{Fore.MAGENTA}https://github.com/jllllll/exllama/releases/latest{Fore.RESET}" + raise Exception( + f"\n{Fore.RED}ExLlama not installed. Get {Fore.CYAN}{pckg}{Fore.RED} from" + f"\n{Fore.MAGENTA}https://github.com/jllllll/exllama/releases/latest{Fore.RESET}" ) except ImportError: - raise ImportError( - f"\n{Fore.RED}Wrong ExLlama version installed. Get {Fore.CYAN}{cu}-{cp}{Fore.RED} from\n" - f"{Fore.MAGENTA}https://github.com/jllllll/exllama/releases/latest{Fore.RESET}" + raise Exception( + f"\n{Fore.RED}Wrong ExLlama wheel installed. Get {Fore.CYAN}{pckg}{Fore.RED} from" + f"\n{Fore.MAGENTA}https://github.com/jllllll/exllama/releases/latest{Fore.RESET}" ) @@ -32,7 +34,6 @@ class Generator: return { "required": { "model": ("GPTQ",), - "lora_dir": ("STRING", {"default": ""}), "stop_on_newline": ([False, True], {"default": False}), "max_tokens": ("INT", {"default": 128, "min": 1, "max": 8192}), "temperature": ("FLOAT", {"default": 0.7, "min": 0.0, "max": 2.0, "step": 0.01}), @@ -43,6 +44,9 @@ class Generator: "seed": ("INT", {"default": 0, "min": 0, "max": 2**64 - 1}), "prompt": ("STRING", {"default": "", "multiline": True}), }, + "optional": { + "lora": ("LORA",), + }, } CATEGORY = "Zuellni/ExLlama" @@ -53,7 +57,6 @@ class Generator: def generate( self, model, - lora_dir, stop_on_newline, max_tokens, temperature, @@ -63,6 +66,7 @@ class Generator: penalty, seed, prompt, + lora=None, ): progress = ProgressBar(max_tokens) prompt = prompt.strip() @@ -77,15 +81,7 @@ class Generator: settings.top_p = top_p settings.typical = typical_p settings.token_repetition_penalty_max = penalty - - if lora_dir: - lora_dir = Path(lora_dir).expanduser() - lora_config = str(lora_dir / "adapter_config.json") - lora_model = str(lora_dir / "adapter_model.bin") - lora = ExLlamaLora(model.model, lora_config, lora_model) - settings.lora = lora - else: - settings.lora = None + settings.lora = lora stop_conditions = [model.tokenizer.eos_token_id] @@ -124,8 +120,6 @@ class Loader: RETURN_TYPES = ("GPTQ",) def load(self, model_dir, max_seq_len): - soft_empty_cache() - model_dir = Path(model_dir).expanduser() config = ExLlamaConfig(str(model_dir / "config.json")) config.model_path = model_dir.glob("*.safetensors") @@ -139,6 +133,29 @@ class Loader: return (generator,) +class Lora: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "model": ("GPTQ",), + "lora_dir": ("STRING", {"default": ""}), + }, + } + + CATEGORY = "Zuellni/ExLlama" + FUNCTION = "load" + RETURN_TYPES = ("LORA",) + + def load(self, model, lora_dir): + lora_dir = Path(lora_dir).expanduser() + lora_config = str(lora_dir / "adapter_config.json") + lora_model = str(lora_dir / "adapter_model.bin") + lora = ExLlamaLora(model.model, lora_config, lora_model) + + return (lora,) + + class Previewer: @classmethod def INPUT_TYPES(cls): @@ -147,7 +164,8 @@ class Previewer: CATEGORY = "Zuellni/ExLlama" FUNCTION = "preview" OUTPUT_NODE = True - RETURN_TYPES = () + RETURN_NAMES = ("TEXT",) + RETURN_TYPES = ("STRING",) def preview(self, text): - return {"ui": {"text": [text]}} + return {"ui": {"text": [text]}, "result": (text,)}