Fix error when model isn't fully unloaded and bump requirements
This commit is contained in:
+1
-1
@@ -63,7 +63,7 @@ class Loader:
|
||||
return (self,)
|
||||
|
||||
def load(self):
|
||||
if self.ckpt:
|
||||
if self.ckpt and self.cache and self.generator:
|
||||
return
|
||||
|
||||
self.ckpt = ExLlamaV2(self.config)
|
||||
|
||||
+2
-2
@@ -1,5 +1,5 @@
|
||||
exllamav2; platform_system == "Linux"
|
||||
https://github.com/turboderp/exllamav2/releases/download/v0.0.9/exllamav2-0.0.9+cu121-cp311-cp311-win_amd64.whl; platform_system == "Windows"
|
||||
https://github.com/turboderp/exllamav2/releases/download/v0.0.10/exllamav2-0.0.10+cu121-cp311-cp311-win_amd64.whl; platform_system == "Windows"
|
||||
|
||||
# flash-attn; platform_system == "Linux"
|
||||
# https://github.com/jllllll/flash-attention/releases/download/v2.3.4/flash_attn-2.3.4+cu121torch2.1cxx11abiFALSE-cp311-cp311-win_amd64.whl; platform_system == "Windows"
|
||||
# https://github.com/jllllll/flash-attention/releases/download/v2.3.6/flash_attn-2.3.6+cu121torch2.1cxx11abiFALSE-cp311-cp311-win_amd64.whl; platform_system == "Windows"
|
||||
|
||||
Reference in New Issue
Block a user