Fix error when model isn't fully unloaded and bump requirements
This commit is contained in:
+1
-1
@@ -63,7 +63,7 @@ class Loader:
|
|||||||
return (self,)
|
return (self,)
|
||||||
|
|
||||||
def load(self):
|
def load(self):
|
||||||
if self.ckpt:
|
if self.ckpt and self.cache and self.generator:
|
||||||
return
|
return
|
||||||
|
|
||||||
self.ckpt = ExLlamaV2(self.config)
|
self.ckpt = ExLlamaV2(self.config)
|
||||||
|
|||||||
+2
-2
@@ -1,5 +1,5 @@
|
|||||||
exllamav2; platform_system == "Linux"
|
exllamav2; platform_system == "Linux"
|
||||||
https://github.com/turboderp/exllamav2/releases/download/v0.0.9/exllamav2-0.0.9+cu121-cp311-cp311-win_amd64.whl; platform_system == "Windows"
|
https://github.com/turboderp/exllamav2/releases/download/v0.0.10/exllamav2-0.0.10+cu121-cp311-cp311-win_amd64.whl; platform_system == "Windows"
|
||||||
|
|
||||||
# flash-attn; platform_system == "Linux"
|
# flash-attn; platform_system == "Linux"
|
||||||
# https://github.com/jllllll/flash-attention/releases/download/v2.3.4/flash_attn-2.3.4+cu121torch2.1cxx11abiFALSE-cp311-cp311-win_amd64.whl; platform_system == "Windows"
|
# https://github.com/jllllll/flash-attention/releases/download/v2.3.6/flash_attn-2.3.6+cu121torch2.1cxx11abiFALSE-cp311-cp311-win_amd64.whl; platform_system == "Windows"
|
||||||
|
|||||||
Reference in New Issue
Block a user