diff --git a/README.md b/README.md index 6d2dd7f..049cf41 100644 --- a/README.md +++ b/README.md @@ -1,4 +1,6 @@ -# [ExLlama](https://github.com/turboderp/exllama) nodes for [ComfyUI](https://github.com/comfyanonymous/ComfyUI). +# ExLlama nodes for ComfyUI +A simple prompt generator for [ComfyUI](https://github.com/comfyanonymous/ComfyUI) utilizing [ExLlama](https://github.com/turboderp/exllama). +Outputs are printed in the console, sadly I have no idea how to append them to metadata or display in the UI. Suggestions welcome. ## Installation Clone the repository to `custom_nodes` in your ComfyUI directory: @@ -6,7 +8,8 @@ Clone the repository to `custom_nodes` in your ComfyUI directory: git clone https://github.com/Zuellni/ComfyUI-ExLlama-Nodes ``` -Install the latest ExLlama package from https://github.com/jllllll/exllama/releases. Choose the version matching your platform, Python, and PyTorch CUDA/ROCm. Example for Windows with Python 3.10 and CUDA 11.7: +Install the latest ExLlama package from https://github.com/jllllll/exllama/releases. Choose the version matching your platform, Python, and PyTorch CUDA/ROCm. +Example for Windows with Python 3.10 and CUDA 11.7: ``` pip install https://github.com/jllllll/exllama/releases/download/0.0.17/exllama-0.0.17+cu117-cp310-cp310-win_amd64.whl ``` @@ -15,12 +18,15 @@ pip install https://github.com/jllllll/exllama/releases/download/0.0.17/exllama- Comes with the following nodes: ### ExLlama Loader -Used to load 4-bit GPTQ Llama/2 models. You can find a lot of them over at https://huggingface.co/TheBloke. ExLlama allocates memory according to `max_seq_len`. Lowering it is a good way to save on GPU RAM. It's not possible to offload the model to CPU RAM currently. +Used to load 4-bit GPTQ Llama/2 models. You can find a lot of them over at https://huggingface.co/TheBloke. +ExLlama allocates memory according to `max_seq_len`. Lowering it is a good way to save on GPU RAM. +It's not possible to offload the model to CPU RAM currently. ### ExLlama Generator -Generates a `string` based on the given `prompt` for use with other nodes. Default parameter values correspond to the `simple-1` preset from https://github.com/oobabooga/text-generation-webui. ExLlama isn't deterministic, so the outputs may differ even with the same seed. +Generates a `string` based on the given `prompt` for use with other nodes. +Default parameter values correspond to the `simple-1` preset from https://github.com/oobabooga/text-generation-webui. +ExLlama isn't deterministic, so the outputs may differ even with the same seed. ## Workflow -![workflow](https://github.com/Zuellni/ComfyUI-ExLlama/assets/123005779/005df502-9986-444c-b736-448b305e329c) - +![workflow](https://github.com/Zuellni/ComfyUI-ExLlama/assets/123005779/005df502-9986-444c-b736-448b305e329c) Can be loaded directly in ComfyUI. diff --git a/nodes.py b/nodes.py index 8530b28..bfdbc4c 100644 --- a/nodes.py +++ b/nodes.py @@ -1,3 +1,4 @@ +import json from pathlib import Path import torch @@ -15,7 +16,7 @@ class Generator: "required": { "model": ("GPTQ",), "tokens": ("INT", {"default": 128, "min": 1, "max": 8192}), - "temp": ("FLOAT", {"default": 0.7, "min": 0.0, "max": 2.0, "step": 0.01}), + "temperature": ("FLOAT", {"default": 0.7, "min": 0.0, "max": 2.0, "step": 0.01}), "top_k": ("INT", {"default": 20, "min": 0, "max": 200}), "top_p": ("FLOAT", {"default": 0.9, "min": 0.0, "max": 1.0, "step": 0.01}), "typical_p": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}), @@ -28,9 +29,10 @@ class Generator: CATEGORY = "Zuellni/ExLlama" FUNCTION = "generate" OUTPUT_NODE = True + RETURN_NAMES = ("TEXT",) RETURN_TYPES = ("STRING",) - def generate(self, model, tokens, temp, top_k, top_p, typical_p, penalty, seed, prompt): + def generate(self, model, tokens, temperature, top_k, top_p, typical_p, penalty, seed, prompt): progress = ProgressBar(tokens) def update(value): @@ -38,7 +40,7 @@ class Generator: progress.update(value) settings = ExLlamaAltGenerator.Settings() - settings.temperature = temp + settings.temperature = temperature settings.top_k = top_k settings.top_p = top_p settings.typical = typical_p @@ -56,6 +58,7 @@ class Generator: output += chunk progress.update(1) + progress.update_absolute(tokens) output = output.strip() print(output) return (output,)