Fix vram usage with higher context
This commit is contained in:
+8
-18
@@ -3,15 +3,8 @@ import random
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from time import time
|
from time import time
|
||||||
|
|
||||||
from exllamav2 import (
|
from exllamav2 import *
|
||||||
ExLlamaV2,
|
from exllamav2.generator import *
|
||||||
ExLlamaV2Cache,
|
|
||||||
ExLlamaV2Cache_8bit,
|
|
||||||
ExLlamaV2Cache_Q4,
|
|
||||||
ExLlamaV2Config,
|
|
||||||
ExLlamaV2Tokenizer,
|
|
||||||
)
|
|
||||||
from exllamav2.generator import ExLlamaV2Sampler, ExLlamaV2StreamingGenerator
|
|
||||||
|
|
||||||
from comfy.model_management import soft_empty_cache, unload_all_models
|
from comfy.model_management import soft_empty_cache, unload_all_models
|
||||||
from comfy.utils import ProgressBar
|
from comfy.utils import ProgressBar
|
||||||
@@ -56,8 +49,10 @@ class Loader:
|
|||||||
|
|
||||||
if max_seq_len:
|
if max_seq_len:
|
||||||
self.config.max_seq_len = max_seq_len
|
self.config.max_seq_len = max_seq_len
|
||||||
self.config.max_input_len = max_seq_len
|
|
||||||
self.config.max_attention_len = max_seq_len**2
|
if self.config.max_input_len > max_seq_len:
|
||||||
|
self.config.max_input_len = max_seq_len
|
||||||
|
self.config.max_attention_len = max_seq_len**2
|
||||||
|
|
||||||
return (self,)
|
return (self,)
|
||||||
|
|
||||||
@@ -85,14 +80,9 @@ class Loader:
|
|||||||
else ExLlamaV2Cache(self.model, lazy=True)
|
else ExLlamaV2Cache(self.model, lazy=True)
|
||||||
)
|
)
|
||||||
|
|
||||||
self.model.load_autosplit(self.cache, callback=lambda _, __: progress.update(1))
|
|
||||||
self.tokenizer = ExLlamaV2Tokenizer(self.config)
|
self.tokenizer = ExLlamaV2Tokenizer(self.config)
|
||||||
|
self.model.load_autosplit(self.cache, callback=lambda _, __: progress.update(1))
|
||||||
self.generator = ExLlamaV2StreamingGenerator(
|
self.generator = ExLlamaV2StreamingGenerator(self.model, self.cache, self.tokenizer)
|
||||||
model=self.model,
|
|
||||||
cache=self.cache,
|
|
||||||
tokenizer=self.tokenizer,
|
|
||||||
)
|
|
||||||
|
|
||||||
def unload(self):
|
def unload(self):
|
||||||
if hasattr(self, "model") and self.model:
|
if hasattr(self, "model") and self.model:
|
||||||
|
|||||||
+1
-1
@@ -1 +1 @@
|
|||||||
exllamav2>=0.0.16; platform_system == "Linux"
|
exllamav2>=0.0.17; platform_system == "Linux"
|
||||||
|
|||||||
Reference in New Issue
Block a user