Garbage collect model before reloading, should fix vram issues

This commit is contained in:
Zuellni
2023-10-07 22:59:45 +02:00
parent aa44ddb7b5
commit ebf8ccab60
+13 -4
View File
@@ -1,6 +1,8 @@
from gc import collect
from time import time from time import time
import torch import torch
from comfy.model_management import soft_empty_cache
from comfy.utils import ProgressBar from comfy.utils import ProgressBar
from exllamav2 import ExLlamaV2, ExLlamaV2Cache, ExLlamaV2Config, ExLlamaV2Tokenizer from exllamav2 import ExLlamaV2, ExLlamaV2Cache, ExLlamaV2Config, ExLlamaV2Tokenizer
from exllamav2.generator import ExLlamaV2Sampler, ExLlamaV2StreamingGenerator from exllamav2.generator import ExLlamaV2Sampler, ExLlamaV2StreamingGenerator
@@ -21,18 +23,25 @@ class Loader:
RETURN_NAMES = ("MODEL",) RETURN_NAMES = ("MODEL",)
RETURN_TYPES = ("EXL_MODEL",) RETURN_TYPES = ("EXL_MODEL",)
def __init__(self):
self.model = None
def load(self, model_dir, max_seq_len): def load(self, model_dir, max_seq_len):
del self.model
collect()
soft_empty_cache()
config = ExLlamaV2Config() config = ExLlamaV2Config()
config.model_dir = model_dir config.model_dir = model_dir
config.prepare() config.prepare()
config.max_seq_len = max_seq_len config.max_seq_len = max_seq_len
model = ExLlamaV2(config) self.model = ExLlamaV2(config)
model.load() self.model.load()
cache = ExLlamaV2Cache(model) cache = ExLlamaV2Cache(self.model)
tokenizer = ExLlamaV2Tokenizer(config) tokenizer = ExLlamaV2Tokenizer(config)
generator = ExLlamaV2StreamingGenerator(model, cache, tokenizer) generator = ExLlamaV2StreamingGenerator(self.model, cache, tokenizer)
settings = ExLlamaV2Sampler.Settings() settings = ExLlamaV2Sampler.Settings()
return ((tokenizer, generator, settings),) return ((tokenizer, generator, settings),)