T5 second GPU offload
This commit is contained in:
@@ -136,6 +136,8 @@ Place them in your `ComfyUI/models/t5` folder. You can put them in a subfolder c
|
||||
|
||||
Loaded onto the CPU, it'll use about 22GBs of system RAM. Depending on which weights you use, it might use slightly more during loading.
|
||||
|
||||
If you have a second GPU, selecting "cuda:1" as the device will allow you to use it for T5, freeing at least some VRAM/System RAM. Using FP16 as the dtype is recommended.
|
||||
|
||||
Loaded in bnb4bit mode, it only takes around 6GB VRAM, making it work with 12GB cards. The only drawback is that it'll constantly stay in VRAM since BitsAndBytes doesn't allow moving the weights to the system RAM temporarily. Switching to a different workflow *should* still release the VRAM as expected. Pascal cards (1080ti, P40) seem to struggle with 4bit. Select "cpu" if you encounter issues.
|
||||
|
||||
On windows, you may need a newer version of bitsandbytes for 4bit. Try `python -m pip install bitsandbytes --prefer-binary --extra-index-url=https://jllllll.github.io/bitsandbytes-windows-webui`
|
||||
|
||||
@@ -35,6 +35,12 @@ class EXM_T5v11:
|
||||
self.load_device = "cpu"
|
||||
self.offload_device = "cpu"
|
||||
self.init_device="cpu"
|
||||
elif device.startswith("cuda"):
|
||||
print("Direct CUDA device override!\nVRAM will not be freed by default.")
|
||||
size = 0
|
||||
self.load_device = device
|
||||
self.offload_device = device
|
||||
self.init_device = device
|
||||
else:
|
||||
size = 0
|
||||
self.load_device = model_management.get_torch_device()
|
||||
|
||||
+5
-1
@@ -33,12 +33,16 @@ else: dtypes += ["FP8 E4M3", "FP8 E5M2"]
|
||||
class T5v11Loader:
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
devices = ["auto", "cpu", "gpu"]
|
||||
# hack for using second GPU as offload
|
||||
for k in range(1, torch.cuda.device_count()):
|
||||
devices.append(f"cuda:{k}")
|
||||
return {
|
||||
"required": {
|
||||
"t5v11_name": (folder_paths.get_filename_list("t5"),),
|
||||
"t5v11_ver": (["xxl"],),
|
||||
"path_type": (["folder", "file"],),
|
||||
"device": (["auto", "cpu", "gpu"],{"default":"cpu"}),
|
||||
"device": (devices, {"default":"cpu"}),
|
||||
"dtype": (dtypes,),
|
||||
}
|
||||
}
|
||||
|
||||
+3
-1
@@ -33,7 +33,9 @@ class T5v11Model(torch.nn.Module):
|
||||
else:
|
||||
if dtype: model_args["torch_dtype"] = dtype
|
||||
self.bnb = False
|
||||
# TODO: custom device map?
|
||||
# second GPU offload hack part 2
|
||||
if device.startswith("cuda"):
|
||||
model_args["device_map"] = device
|
||||
print(f"Loading T5 from '{textmodel_path}'")
|
||||
self.transformer = T5EncoderModel.from_pretrained(textmodel_path, **model_args)
|
||||
else:
|
||||
|
||||
Reference in New Issue
Block a user