From e2905be4c33e29d485753492617f968b86ea4006 Mon Sep 17 00:00:00 2001 From: kijai <40791699+kijai@users.noreply.github.com> Date: Fri, 9 Aug 2024 18:36:11 +0300 Subject: [PATCH] Update nodes.py --- nodes.py | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/nodes.py b/nodes.py index 7a5faef..4ee6045 100644 --- a/nodes.py +++ b/nodes.py @@ -31,6 +31,11 @@ class DownloadAndLoadLLaVAOneVisionModel: { "default": 'fp16' }), + "attention": ( + [ 'flash_attention_2', 'sdpa', 'eager'], + { + "default": 'sdpa' + }), }, } @@ -40,12 +45,13 @@ class DownloadAndLoadLLaVAOneVisionModel: FUNCTION = "loadmodel" CATEGORY = "LLaVA-OneVision" - def loadmodel(self, model, device, precision): + def loadmodel(self, model, device, precision, attention): if precision != 'fp32' and device == 'cpu': raise ValueError("fp16 and bf16 are not supported on cpu") dtype = {"bf16": torch.bfloat16, "fp16": torch.float16, "fp32": torch.float32}[precision] device = {"cuda": torch.device("cuda"), "cpu": torch.device("cpu"), "mps": torch.device("mps")}[device] + print(f"using {attention} for attention") model_name = model.split('/')[-1] download_path = os.path.join(folder_paths.models_dir, "LLM", "LLaVA-OneVision", model_name) @@ -63,6 +69,7 @@ class DownloadAndLoadLLaVAOneVisionModel: model, None, model_name="llava_qwen", + attn_implementation=attention, load_8bit=False, load_4bit=False )