Switch to dynamic generator
This commit is contained in:
+21
-17
@@ -31,7 +31,7 @@ class Loader:
|
|||||||
return {
|
return {
|
||||||
"required": {
|
"required": {
|
||||||
"model": (models, {"default": default}),
|
"model": (models, {"default": default}),
|
||||||
"cache_bits": ((4, 8, 16), {"default": 16}),
|
"cache_bits": ((4, 6, 8, 16), {"default": 16}),
|
||||||
"max_seq_len": ("INT", {"default": 2048, "max": 2**20}),
|
"max_seq_len": ("INT", {"default": 2048, "max": 2**20}),
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
@@ -75,14 +75,16 @@ class Loader:
|
|||||||
self.cache = (
|
self.cache = (
|
||||||
ExLlamaV2Cache_Q4(self.model, lazy=True)
|
ExLlamaV2Cache_Q4(self.model, lazy=True)
|
||||||
if self.cache_bits == 4
|
if self.cache_bits == 4
|
||||||
else ExLlamaV2Cache_8bit(self.model, lazy=True)
|
else ExLlamaV2Cache_Q6(self.model, lazy=True)
|
||||||
|
if self.cache_bits == 6
|
||||||
|
else ExLlamaV2Cache_Q8(self.model, lazy=True)
|
||||||
if self.cache_bits == 8
|
if self.cache_bits == 8
|
||||||
else ExLlamaV2Cache(self.model, lazy=True)
|
else ExLlamaV2Cache(self.model, lazy=True)
|
||||||
)
|
)
|
||||||
|
|
||||||
self.tokenizer = ExLlamaV2Tokenizer(self.config)
|
self.tokenizer = ExLlamaV2Tokenizer(self.config)
|
||||||
self.model.load_autosplit(self.cache, callback=lambda _, __: progress.update(1))
|
self.model.load_autosplit(self.cache, callback=lambda _, __: progress.update(1))
|
||||||
self.generator = ExLlamaV2StreamingGenerator(self.model, self.cache, self.tokenizer)
|
self.generator = ExLlamaV2DynamicGenerator(self.model, self.cache, self.tokenizer)
|
||||||
|
|
||||||
def unload(self):
|
def unload(self):
|
||||||
if hasattr(self, "model") and self.model:
|
if hasattr(self, "model") and self.model:
|
||||||
@@ -155,6 +157,7 @@ class Generator:
|
|||||||
model.unload()
|
model.unload()
|
||||||
|
|
||||||
model.load()
|
model.load()
|
||||||
|
random.seed(seed)
|
||||||
input = model.tokenizer.encode(text, encode_special_tokens=True)
|
input = model.tokenizer.encode(text, encode_special_tokens=True)
|
||||||
input_len = input.shape[-1]
|
input_len = input.shape[-1]
|
||||||
max_len = model.config.max_seq_len - input_len
|
max_len = model.config.max_seq_len - input_len
|
||||||
@@ -164,10 +167,7 @@ class Generator:
|
|||||||
max_tokens = max_len
|
max_tokens = max_len
|
||||||
|
|
||||||
if single_line:
|
if single_line:
|
||||||
stop.append(model.tokenizer.newline_token_id)
|
stop.append("\n")
|
||||||
|
|
||||||
model.generator.set_stop_conditions(stop)
|
|
||||||
random.seed(seed)
|
|
||||||
|
|
||||||
settings = ExLlamaV2Sampler.Settings()
|
settings = ExLlamaV2Sampler.Settings()
|
||||||
settings.temperature = temperature
|
settings.temperature = temperature
|
||||||
@@ -179,21 +179,25 @@ class Generator:
|
|||||||
settings.token_repetition_penalty = repetition_penalty
|
settings.token_repetition_penalty = repetition_penalty
|
||||||
settings.temperature_last = temperature_last
|
settings.temperature_last = temperature_last
|
||||||
|
|
||||||
start = time()
|
job = ExLlamaV2DynamicJob(input, max_new_tokens=max_tokens, stop_conditions=stop)
|
||||||
model.generator.begin_stream_ex(input, settings)
|
model.generator.enqueue(job)
|
||||||
|
|
||||||
progress = ProgressBar(max_tokens)
|
progress = ProgressBar(max_tokens)
|
||||||
|
start = time()
|
||||||
|
|
||||||
eos = False
|
eos = False
|
||||||
output = ""
|
chunks = []
|
||||||
tokens = 0
|
tokens = 0
|
||||||
|
|
||||||
while not eos and tokens < max_tokens:
|
while not eos:
|
||||||
response = model.generator.stream_ex()
|
for response in model.generator.iterate():
|
||||||
output += response["chunk"]
|
if response["stage"] == "streaming":
|
||||||
eos = response["eos"]
|
chunks.append(response.get("text", ""))
|
||||||
progress.update(1)
|
eos = response["eos"]
|
||||||
tokens += 1
|
progress.update(1)
|
||||||
|
tokens += 1
|
||||||
|
|
||||||
output = output.strip()
|
output = "".join(chunks).strip()
|
||||||
total = round(time() - start, 2)
|
total = round(time() - start, 2)
|
||||||
speed = round(tokens / total, 2)
|
speed = round(tokens / total, 2)
|
||||||
|
|
||||||
|
|||||||
+2
-1
@@ -1 +1,2 @@
|
|||||||
exllamav2>=0.0.17; platform_system == "Linux"
|
exllamav2; platform_system == "Linux"
|
||||||
|
flash-attn; platform_system == "Linux"
|
||||||
|
|||||||
Reference in New Issue
Block a user