Update progress after gen is finished

This commit is contained in:
Zuellni
2023-09-13 16:23:00 +02:00
parent 55bbfd4606
commit 5e1fc0de87
2 changed files with 18 additions and 9 deletions
+12 -6
View File
@@ -1,4 +1,6 @@
# [ExLlama](https://github.com/turboderp/exllama) nodes for [ComfyUI](https://github.com/comfyanonymous/ComfyUI).
# ExLlama nodes for ComfyUI
A simple prompt generator for [ComfyUI](https://github.com/comfyanonymous/ComfyUI) utilizing [ExLlama](https://github.com/turboderp/exllama).
Outputs are printed in the console, sadly I have no idea how to append them to metadata or display in the UI. Suggestions welcome.
## Installation
Clone the repository to `custom_nodes` in your ComfyUI directory:
@@ -6,7 +8,8 @@ Clone the repository to `custom_nodes` in your ComfyUI directory:
git clone https://github.com/Zuellni/ComfyUI-ExLlama-Nodes
```
Install the latest ExLlama package from https://github.com/jllllll/exllama/releases. Choose the version matching your platform, Python, and PyTorch CUDA/ROCm. Example for Windows with Python 3.10 and CUDA 11.7:
Install the latest ExLlama package from https://github.com/jllllll/exllama/releases. Choose the version matching your platform, Python, and PyTorch CUDA/ROCm.
Example for Windows with Python 3.10 and CUDA 11.7:
```
pip install https://github.com/jllllll/exllama/releases/download/0.0.17/exllama-0.0.17+cu117-cp310-cp310-win_amd64.whl
```
@@ -15,12 +18,15 @@ pip install https://github.com/jllllll/exllama/releases/download/0.0.17/exllama-
Comes with the following nodes:
### ExLlama Loader
Used to load 4-bit GPTQ Llama/2 models. You can find a lot of them over at https://huggingface.co/TheBloke. ExLlama allocates memory according to `max_seq_len`. Lowering it is a good way to save on GPU RAM. It's not possible to offload the model to CPU RAM currently.
Used to load 4-bit GPTQ Llama/2 models. You can find a lot of them over at https://huggingface.co/TheBloke.
ExLlama allocates memory according to `max_seq_len`. Lowering it is a good way to save on GPU RAM.
It's not possible to offload the model to CPU RAM currently.
### ExLlama Generator
Generates a `string` based on the given `prompt` for use with other nodes. Default parameter values correspond to the `simple-1` preset from https://github.com/oobabooga/text-generation-webui. ExLlama isn't deterministic, so the outputs may differ even with the same seed.
Generates a `string` based on the given `prompt` for use with other nodes.
Default parameter values correspond to the `simple-1` preset from https://github.com/oobabooga/text-generation-webui.
ExLlama isn't deterministic, so the outputs may differ even with the same seed.
## Workflow
![workflow](https://github.com/Zuellni/ComfyUI-ExLlama/assets/123005779/005df502-9986-444c-b736-448b305e329c)
![workflow](https://github.com/Zuellni/ComfyUI-ExLlama/assets/123005779/005df502-9986-444c-b736-448b305e329c)
Can be loaded directly in ComfyUI.
+6 -3
View File
@@ -1,3 +1,4 @@
import json
from pathlib import Path
import torch
@@ -15,7 +16,7 @@ class Generator:
"required": {
"model": ("GPTQ",),
"tokens": ("INT", {"default": 128, "min": 1, "max": 8192}),
"temp": ("FLOAT", {"default": 0.7, "min": 0.0, "max": 2.0, "step": 0.01}),
"temperature": ("FLOAT", {"default": 0.7, "min": 0.0, "max": 2.0, "step": 0.01}),
"top_k": ("INT", {"default": 20, "min": 0, "max": 200}),
"top_p": ("FLOAT", {"default": 0.9, "min": 0.0, "max": 1.0, "step": 0.01}),
"typical_p": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}),
@@ -28,9 +29,10 @@ class Generator:
CATEGORY = "Zuellni/ExLlama"
FUNCTION = "generate"
OUTPUT_NODE = True
RETURN_NAMES = ("TEXT",)
RETURN_TYPES = ("STRING",)
def generate(self, model, tokens, temp, top_k, top_p, typical_p, penalty, seed, prompt):
def generate(self, model, tokens, temperature, top_k, top_p, typical_p, penalty, seed, prompt):
progress = ProgressBar(tokens)
def update(value):
@@ -38,7 +40,7 @@ class Generator:
progress.update(value)
settings = ExLlamaAltGenerator.Settings()
settings.temperature = temp
settings.temperature = temperature
settings.top_k = top_k
settings.top_p = top_p
settings.typical = typical_p
@@ -56,6 +58,7 @@ class Generator:
output += chunk
progress.update(1)
progress.update_absolute(tokens)
output = output.strip()
print(output)
return (output,)