From ebf8ccab60546ebd4a9a6ace8f63d50c91f3a845 Mon Sep 17 00:00:00 2001 From: Zuellni <123005779+Zuellni@users.noreply.github.com> Date: Sat, 7 Oct 2023 22:59:45 +0200 Subject: [PATCH] Garbage collect model before reloading, should fix vram issues --- exllama.py | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/exllama.py b/exllama.py index 605d18d..961407a 100644 --- a/exllama.py +++ b/exllama.py @@ -1,6 +1,8 @@ +from gc import collect from time import time import torch +from comfy.model_management import soft_empty_cache from comfy.utils import ProgressBar from exllamav2 import ExLlamaV2, ExLlamaV2Cache, ExLlamaV2Config, ExLlamaV2Tokenizer from exllamav2.generator import ExLlamaV2Sampler, ExLlamaV2StreamingGenerator @@ -21,18 +23,25 @@ class Loader: RETURN_NAMES = ("MODEL",) RETURN_TYPES = ("EXL_MODEL",) + def __init__(self): + self.model = None + def load(self, model_dir, max_seq_len): + del self.model + collect() + soft_empty_cache() + config = ExLlamaV2Config() config.model_dir = model_dir config.prepare() config.max_seq_len = max_seq_len - model = ExLlamaV2(config) - model.load() + self.model = ExLlamaV2(config) + self.model.load() - cache = ExLlamaV2Cache(model) + cache = ExLlamaV2Cache(self.model) tokenizer = ExLlamaV2Tokenizer(config) - generator = ExLlamaV2StreamingGenerator(model, cache, tokenizer) + generator = ExLlamaV2StreamingGenerator(self.model, cache, tokenizer) settings = ExLlamaV2Sampler.Settings() return ((tokenizer, generator, settings),)